rspecq-instructure 0.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,634 @@
1
+ require "redis"
2
+
3
+ module RSpecQ
4
+ # Queue is the data store interface (Redis) and is used to manage the work
5
+ # queue for a particular build. All Redis operations happen via Queue.
6
+ #
7
+ # A queue typically contains all the data needed for a particular build to
8
+ # happen. These include (but are not limited to) the following:
9
+ #
10
+ # - the list of jobs (spec files and/or examples) to be executed
11
+ # - the failed examples along with their backtrace
12
+ # - the set of running jobs
13
+ # - previous job timing statistics used to optimally schedule the jobs
14
+ # - the set of executed jobs
15
+ class Queue
16
+ RESERVE_JOB = <<~LUA.freeze
17
+ local queue = KEYS[1]
18
+ local queue_running = KEYS[2]
19
+ local worker_id = ARGV[1]
20
+
21
+ local job = redis.call('lpop', queue)
22
+ if job then
23
+ redis.call('hset', queue_running, worker_id, job)
24
+ return job
25
+ else
26
+ return nil
27
+ end
28
+ LUA
29
+
30
+ # Scans for dead workers and puts their reserved jobs back to the queue.
31
+ REQUEUE_LOST_JOB = <<~LUA.freeze
32
+ local worker_heartbeats = KEYS[1]
33
+ local queue_running = KEYS[2]
34
+ local queue_unprocessed = KEYS[3]
35
+ local queue_lost = KEYS[4]
36
+ local time_now = ARGV[1]
37
+ local timeout = ARGV[2]
38
+
39
+ local dead_workers = redis.call('zrangebyscore', worker_heartbeats, 0, time_now - timeout)
40
+ for _, worker in ipairs(dead_workers) do
41
+ local job = redis.call('hget', queue_running, worker)
42
+ if job then
43
+ redis.call('lpush', queue_unprocessed, job)
44
+ redis.call('hdel', queue_running, worker)
45
+ redis.call('zincrby', queue_lost, 1, job)
46
+
47
+ return {job, worker}
48
+ end
49
+ end
50
+
51
+ return nil
52
+ LUA
53
+
54
+ REQUEUE_JOB = <<~LUA.freeze
55
+ local key_queue_unprocessed = KEYS[1]
56
+ local key_requeues = KEYS[2]
57
+ local key_requeued_job_original_worker = KEYS[3]
58
+ local key_job_location = KEYS[4]
59
+ local job = ARGV[1]
60
+ local max_requeues = ARGV[2]
61
+ local original_worker = ARGV[3]
62
+ local location = ARGV[4]
63
+
64
+ local requeued_times = redis.call('hget', key_requeues, job)
65
+ if requeued_times and tonumber(requeued_times) >= tonumber(max_requeues) then
66
+ return nil
67
+ end
68
+
69
+ redis.call('lpush', key_queue_unprocessed, job)
70
+ redis.call('hset', key_requeued_job_original_worker, job, original_worker)
71
+ redis.call('hincrby', key_requeues, job, 1)
72
+ redis.call('hset', key_job_location, job, location)
73
+
74
+ return true
75
+ LUA
76
+
77
+ STATUS_INITIALIZING = "initializing".freeze
78
+ STATUS_READY = "ready".freeze
79
+ STATUS_SUCCESS = "success".freeze
80
+ STATUS_FAILURE = "failure".freeze
81
+
82
+ # Per-build timings are only needed until the reporter promotes them to the
83
+ # global key; expire them so always-on recording can't grow Redis unbounded.
84
+ # Interim measure until comprehensive key TTLs land (DE-1805).
85
+ BUILD_TIMINGS_TTL_SEC = 86_400
86
+
87
+ attr_reader :redis
88
+
89
+ def initialize(build_id, worker_id, redis_opts, worker_liveness_sec)
90
+ @build_id = build_id
91
+ @worker_id = worker_id
92
+ @redis = Redis.new(redis_opts.merge(id: worker_id))
93
+ @worker_liveness_sec = worker_liveness_sec
94
+ end
95
+
96
+ # The build's final status once finished: STATUS_SUCCESS or STATUS_FAILURE
97
+ # (or the lifecycle STATUS_INITIALIZING / STATUS_READY before then).
98
+ def status
99
+ @redis.get(key_queue_status)
100
+ end
101
+
102
+ # NOTE: jobs will be processed from head to tail (lpop)
103
+ def publish(jobs, fail_fast = 0)
104
+ time = current_time
105
+ @redis.multi do |pipeline|
106
+ pipeline.hset(key_queue_config, "fail_fast", fail_fast)
107
+ pipeline.rpush(key_queue_unprocessed, jobs)
108
+ pipeline.setnx(key_queue_ready_at, time)
109
+ pipeline.set(key_queue_status, STATUS_READY)
110
+ end
111
+
112
+ jobs.size
113
+ end
114
+
115
+ # Records when a master worker was elected (start of the whole build).
116
+ def mark_elected_master_at
117
+ @redis.set(key_elected_master_at, current_time)
118
+ end
119
+
120
+ TRY_MARK_FINISHED = <<~LUA.freeze
121
+ local key_queue_finished_at = KEYS[1]
122
+ local key_queue_status = KEYS[2]
123
+ local key_failures = KEYS[3]
124
+ local key_errors = KEYS[4]
125
+ local key_queue_unprocessed = KEYS[5]
126
+ local key_queue_running = KEYS[6]
127
+ local key_queue_config = KEYS[7]
128
+ local status_success = ARGV[1]
129
+ local status_failure = ARGV[2]
130
+
131
+ local unprocessed_count = redis.call('llen', key_queue_unprocessed)
132
+ local running_count = redis.call('hlen', key_queue_running)
133
+ local failures_count = redis.call('hlen', key_failures)
134
+ local errors_count = redis.call('hlen', key_errors)
135
+ local fail_fast = tonumber(redis.call('hget', key_queue_config, 'fail_fast'))
136
+
137
+ local is_fail_fast = fail_fast and fail_fast > 0 and failures_count + errors_count >= fail_fast
138
+ local is_exhausted = unprocessed_count + running_count == 0
139
+
140
+ if not is_fail_fast and not is_exhausted then
141
+ return nil
142
+ end
143
+
144
+ local current_time = redis.call('time')[1]
145
+
146
+ -- setnx acts as the lock: only the first caller marks the build finished
147
+ local locked = redis.call('setnx', key_queue_finished_at, current_time)
148
+ if locked == 0 then
149
+ return nil
150
+ end
151
+
152
+ if is_fail_fast then
153
+ redis.call('set', key_queue_status, status_failure)
154
+ elseif failures_count + errors_count == 0 then
155
+ redis.call('set', key_queue_status, status_success)
156
+ else
157
+ redis.call('set', key_queue_status, status_failure)
158
+ end
159
+
160
+ return true
161
+ LUA
162
+
163
+ # Marks the build finished (setnx lock, first caller wins) and stores the
164
+ # final status (success/failure). Defensively re-checks the build is over.
165
+ def try_mark_finished
166
+ eval_script(
167
+ TRY_MARK_FINISHED,
168
+ keys: [
169
+ key_queue_finished_at,
170
+ key_queue_status,
171
+ key_failures,
172
+ key_errors,
173
+ key_queue_unprocessed,
174
+ key_queue_running,
175
+ key_queue_config
176
+ ],
177
+ argv: [STATUS_SUCCESS, STATUS_FAILURE]
178
+ )
179
+ end
180
+
181
+ # [seconds from master election, seconds from queue ready] to finish, or
182
+ # nil if the build has not both started and finished.
183
+ def took_times_secs
184
+ elected_master_at = @redis.get(key_elected_master_at)
185
+ ready_at = @redis.get(key_queue_ready_at)
186
+ finished_at = @redis.get(key_queue_finished_at)
187
+
188
+ return nil if elected_master_at.nil? || ready_at.nil? || finished_at.nil?
189
+
190
+ [
191
+ finished_at.to_i - elected_master_at.to_i,
192
+ finished_at.to_i - ready_at.to_i
193
+ ]
194
+ end
195
+
196
+ def reserve_job
197
+ eval_script(
198
+ RESERVE_JOB,
199
+ keys: [
200
+ key_queue_unprocessed,
201
+ key_queue_running,
202
+ ],
203
+ argv: [@worker_id]
204
+ )
205
+ end
206
+
207
+ # If this worker has a job in the running hash (from a previous crash),
208
+ # put it back on the queue. This must be called before update_heartbeat
209
+ # or reserve_job when a worker restarts with the same worker_id.
210
+ def recover_own_job
211
+ job = @redis.hget(key_queue_running, @worker_id)
212
+ return nil unless job
213
+
214
+ @redis.multi do |pipeline|
215
+ pipeline.lpush(key_queue_unprocessed, job)
216
+ pipeline.hdel(key_queue_running, @worker_id)
217
+ end
218
+ job
219
+ end
220
+
221
+ def requeue_lost_job
222
+ eval_script(
223
+ REQUEUE_LOST_JOB,
224
+ keys: [
225
+ key_worker_heartbeats,
226
+ key_queue_running,
227
+ key_queue_unprocessed,
228
+ key_queue_lost
229
+ ],
230
+ argv: [
231
+ current_time,
232
+ @worker_liveness_sec
233
+ ]
234
+ )
235
+ end
236
+
237
+ # Number of unique jobs that were lost and requeued (e.g. by abnormal
238
+ # worker termination). A job could be lost more than once (unlikely).
239
+ def lost_jobs_count
240
+ @redis.zcard(key_queue_lost)
241
+ end
242
+
243
+ # NOTE: The same job might happen to be acknowledged more than once, in
244
+ # the case of requeues.
245
+ def acknowledge_job(job)
246
+ @redis.multi do |pipeline|
247
+ pipeline.hdel(key_queue_running, @worker_id)
248
+ pipeline.sadd(key_queue_processed, job)
249
+ pipeline.rpush(key("queue", "jobs_per_worker", @worker_id), job)
250
+ end
251
+ end
252
+
253
+ # Put job at the head of the queue to be re-processed right after, by
254
+ # another worker. This is a mitigation measure against flaky tests.
255
+ #
256
+ # Returns nil if the job hit the requeue limit and therefore was not
257
+ # requeued and should be considered a failure.
258
+ def requeue_job(example, max_requeues, original_worker_id)
259
+ return false if max_requeues.zero?
260
+
261
+ job = example.id
262
+ location = example.location_rerun_argument
263
+
264
+ eval_script(
265
+ REQUEUE_JOB,
266
+ keys: [key_queue_unprocessed, key_requeues, key("requeued_job_original_worker"), key("job_location")],
267
+ argv: [job, max_requeues, original_worker_id, location]
268
+ )
269
+ end
270
+
271
+ def save_worker_seed(worker, seed)
272
+ @redis.hset(key("worker_seed"), worker, seed)
273
+ end
274
+
275
+ def job_location(job)
276
+ @redis.hget(key("job_location"), job)
277
+ end
278
+
279
+ def is_requeue(job)
280
+ @redis.hget(key_requeues, job)
281
+ end
282
+
283
+ def failed_job_worker(job)
284
+ redis.hget(key("requeued_job_original_worker"), job)
285
+ end
286
+
287
+ def job_rerun_command(job)
288
+ worker = failed_job_worker(job)
289
+ jobs = redis.lrange(key("queue", "jobs_per_worker", worker), 0, -1)
290
+ # Get the job index or (||) the file index incase we queued the entire file
291
+ # or get all the worker jobs incase something has gone VERY wrong
292
+ job_index = jobs.find_index(job) || jobs.find_index(job.split("[")[0]) || -1
293
+ seed = redis.hget(key("worker_seed"), worker)
294
+
295
+ "DISABLE_SPRING=1 DISABLE_BOOTSNAP=1 bin/rspecq --build 1 " \
296
+ "--worker foo --seed #{seed} --max-requeues 0 --fail-fast 1 " \
297
+ "--reproduction #{jobs[0..job_index].join(' ')}"
298
+ end
299
+
300
+ def record_example_failure(example_id, message)
301
+ @redis.hset(key_failures, example_id, message)
302
+ end
303
+
304
+ def record_flaky_failure(example_id, message)
305
+ @redis.hset(key_flaky_failures, example_id, message)
306
+ end
307
+
308
+ # For errors occured outside of examples (e.g. while loading a spec file)
309
+ def record_non_example_error(job, message)
310
+ @redis.hset(key_errors, job, message)
311
+ end
312
+
313
+ # Records a job's timing into the per-build timings key (promoted to the
314
+ # global key by the reporter when --update-timings is set). Also accumulates
315
+ # total worker execution time for the build.
316
+ def record_build_timing(job, duration)
317
+ @redis.pipelined do |pipeline|
318
+ pipeline.zadd(key_build_timings, duration, job)
319
+ pipeline.incrby(key_build_execution_time_ms, (duration * 1000).to_i)
320
+ pipeline.expire(key_build_timings, BUILD_TIMINGS_TTL_SEC)
321
+ pipeline.expire(key_build_execution_time_ms, BUILD_TIMINGS_TTL_SEC)
322
+ end
323
+ end
324
+
325
+ # This build's recorded duration for a single job (seconds), or nil.
326
+ def job_build_timing(job)
327
+ @redis.zscore(key_build_timings, job)
328
+ end
329
+
330
+ # Total worker execution time (sum of all job durations) for this build.
331
+ def total_execution_time_ms
332
+ Integer(@redis.get(key_build_execution_time_ms) || 0)
333
+ end
334
+
335
+ # Promotes this build's timings to the global (or a caller-specified) key.
336
+ # PERSIST clears the TTL that COPY inherits from the build-scoped source
337
+ # key; the global timings key is the durable scheduling basis and must not
338
+ # expire between --update-timings builds.
339
+ def update_global_timings(dst = key_timings)
340
+ @redis.copy(key_build_timings, dst, replace: true)
341
+ @redis.persist(dst)
342
+ end
343
+
344
+ def record_build_time(duration)
345
+ @redis.multi do |pipeline|
346
+ pipeline.lpush(key_build_times, Float(duration))
347
+ pipeline.ltrim(key_build_times, 0, 99)
348
+ pipeline.set(key_build_time, Integer(duration * 1000))
349
+ end
350
+ end
351
+
352
+ def record_worker_heartbeat
353
+ @redis.zadd(key_worker_heartbeats, current_time, @worker_id)
354
+ end
355
+
356
+ def increment_example_count(n)
357
+ @redis.incrby(key_example_count, n)
358
+ end
359
+
360
+ def example_count
361
+ @redis.get(key_example_count).to_i
362
+ end
363
+
364
+ def processed_jobs_count
365
+ @redis.scard(key_queue_processed)
366
+ end
367
+
368
+ def processed_jobs
369
+ @redis.smembers(key_queue_processed)
370
+ end
371
+
372
+ def requeued_jobs
373
+ @redis.hgetall(key_requeues).transform_values(&:to_i)
374
+ end
375
+
376
+ def become_master
377
+ @redis.setnx(key_queue_status, STATUS_INITIALIZING)
378
+ end
379
+
380
+ # Global timings for scheduling, ordered by execution time desc. Whole-file
381
+ # timings are reconstructed from any per-example ("file[...]") entries so the
382
+ # scheduler can still recognize a split file as slow and re-split it.
383
+ def global_timings
384
+ redis_timings = @redis.zrevrange(key_timings, 0, -1, withscores: true).to_h
385
+
386
+ whole_file_timings = populate_splitted_file_timings(redis_timings)
387
+ return redis_timings if whole_file_timings.empty?
388
+
389
+ # Real (stored) timings win over reconstructed sums, so a genuine
390
+ # whole-file run is not overridden by a partial (e.g. requeue) sum.
391
+ whole_file_timings.merge!(redis_timings)
392
+ whole_file_timings.sort_by { |_j, d| -d }.to_h
393
+ end
394
+
395
+ def example_failures
396
+ @redis.hgetall(key_failures)
397
+ end
398
+
399
+ def flaky_failures
400
+ @redis.hgetall(key_flaky_failures)
401
+ end
402
+
403
+ def non_example_errors
404
+ @redis.hgetall(key_errors)
405
+ end
406
+
407
+ # True if the build is complete, false otherwise
408
+ def exhausted?
409
+ return false if !published?
410
+
411
+ @redis.multi do |pipeline|
412
+ pipeline.llen(key_queue_unprocessed)
413
+ pipeline.hlen(key_queue_running)
414
+ end.sum.zero?
415
+ end
416
+
417
+ def published?
418
+ [STATUS_READY, STATUS_SUCCESS, STATUS_FAILURE].include?(@redis.get(key_queue_status))
419
+ end
420
+
421
+ def wait_until_published(timeout = 30)
422
+ (timeout * 10).times do
423
+ return if published?
424
+
425
+ sleep 0.1
426
+ end
427
+
428
+ raise "Queue not yet published after #{timeout} seconds"
429
+ end
430
+
431
+ def build_successful?
432
+ exhausted? && example_failures.empty? && non_example_errors.empty?
433
+ end
434
+
435
+ # The remaining jobs to be processed. Jobs at the head of the list will
436
+ # be procesed first.
437
+ def unprocessed_jobs
438
+ @redis.lrange(key_queue_unprocessed, 0, -1)
439
+ end
440
+
441
+ # Returns the jobs considered flaky (i.e. initially failed but passed
442
+ # after being retried). Must be called after the build is complete,
443
+ # otherwise an exception will be raised.
444
+ def flaky_jobs
445
+ if !exhausted? && !build_failed_fast?
446
+ raise "Queue is not yet exhausted"
447
+ end
448
+
449
+ requeued = @redis.hkeys(key_requeues)
450
+
451
+ return [] if requeued.empty?
452
+
453
+ requeued - @redis.hkeys(key_failures)
454
+ end
455
+
456
+ # Returns the number of failures that will trigger the build to fail-fast.
457
+ # Returns 0 if this feature is disabled and nil if the Queue is not yet
458
+ # published
459
+ def fail_fast
460
+ return nil unless published?
461
+
462
+ @fail_fast ||= Integer(@redis.hget(key_queue_config, "fail_fast"))
463
+ end
464
+
465
+ # Returns true if the number of failed tests, has surpassed the threshold
466
+ # to render the run unsuccessful and the build should be terminated.
467
+ def build_failed_fast?
468
+ if fail_fast.nil? || fail_fast.zero?
469
+ return false
470
+ end
471
+
472
+ @redis.multi do |pipeline|
473
+ pipeline.hlen(key_failures)
474
+ pipeline.hlen(key_errors)
475
+ end.sum >= fail_fast
476
+ end
477
+
478
+ # redis: STRING [STATUS_INITIALIZING, STATUS_READY, STATUS_SUCCESS, STATUS_FAILURE]
479
+ def key_queue_status
480
+ key("queue", "status")
481
+ end
482
+
483
+ # redis: HASH<config_key => config_value>
484
+ def key_queue_config
485
+ key("queue", "config")
486
+ end
487
+
488
+ # redis: LIST<job>
489
+ def key_queue_unprocessed
490
+ key("queue", "unprocessed")
491
+ end
492
+
493
+ # redis: HASH<worker_id => job>
494
+ def key_queue_running
495
+ key("queue", "running")
496
+ end
497
+
498
+ # redis: SET<job>
499
+ def key_queue_processed
500
+ key("queue", "processed")
501
+ end
502
+
503
+ # redis: STRING<timestamp> — when a master worker was elected.
504
+ def key_elected_master_at
505
+ key("queue", "elected_master_at")
506
+ end
507
+
508
+ # redis: STRING<timestamp> — when the queue was published (ready).
509
+ def key_queue_ready_at
510
+ key("queue", "ready_at")
511
+ end
512
+
513
+ # redis: STRING<timestamp> — when the build finished (first worker to see
514
+ # the queue exhausted, or fail-fast).
515
+ def key_queue_finished_at
516
+ key("queue", "finished_at")
517
+ end
518
+
519
+ # redis: ZSET<job => times_lost>
520
+ def key_queue_lost
521
+ key("queue", "lost")
522
+ end
523
+
524
+ # Contains regular RSpec example failures.
525
+ #
526
+ # redis: HASH<example_id => error message>
527
+ def key_failures
528
+ key("example_failures")
529
+ end
530
+
531
+ # Contains flaky RSpec example failures.
532
+ #
533
+ # redis: HASH<example_id => error message>
534
+ def key_flaky_failures
535
+ key("flaky_failures")
536
+ end
537
+
538
+ # Contains errors raised outside of RSpec examples
539
+ # (e.g. a syntax error in spec_helper.rb).
540
+ #
541
+ # redis: HASH<job => error message>
542
+ def key_errors
543
+ key("errors")
544
+ end
545
+
546
+ # As a mitigation mechanism for flaky tests, we requeue example failures
547
+ # to be retried by another worker, up to a certain number of times.
548
+ #
549
+ # redis: HASH<job => times_retried>
550
+ def key_requeues
551
+ key("requeues")
552
+ end
553
+
554
+ # The total number of examples, those that were requeued.
555
+ #
556
+ # redis: STRING<integer>
557
+ def key_example_count
558
+ key("example_count")
559
+ end
560
+
561
+ # redis: ZSET<worker_id => timestamp>
562
+ #
563
+ # Timestamp of the last example processed by each worker.
564
+ def key_worker_heartbeats
565
+ key("worker_heartbeats")
566
+ end
567
+
568
+ # redis: ZSET<job => duration>
569
+ #
570
+ # NOTE: This key is not scoped to a build (i.e. shared among all builds),
571
+ # so be careful to only publish timings from a single branch (e.g. master).
572
+ # Otherwise, timings won't be accurate.
573
+ def key_timings
574
+ "timings"
575
+ end
576
+
577
+ # redis: ZSET<job => duration>, scoped to this build. Promoted to the global
578
+ # key_timings by the reporter when --update-timings is set.
579
+ def key_build_timings
580
+ key("timings")
581
+ end
582
+
583
+ # redis: STRING<ms> — total worker execution time for this build.
584
+ def key_build_execution_time_ms
585
+ key("build_execution_time_ms")
586
+ end
587
+
588
+ # redis: LIST<duration>
589
+ #
590
+ # Last build is at the head of the list.
591
+ def key_build_times
592
+ "build_times"
593
+ end
594
+
595
+ def key_build_time
596
+ key("build_time")
597
+ end
598
+
599
+ private
600
+
601
+ # Plain EVAL (not evalsha): canvas fronts Redis with a Twemproxy
602
+ # compatibility guard that forbids SCRIPT LOAD (which evalsha requires),
603
+ # while EVAL is allowed. EVAL also lets a shared/proxied Redis stay
604
+ # scriptless-cache-agnostic.
605
+ def eval_script(script, keys: [], argv: [])
606
+ @redis.eval(script, keys: keys, argv: argv)
607
+ end
608
+
609
+ def key(*keys)
610
+ [@build_id, keys].join(":")
611
+ end
612
+
613
+ # We don't use any Ruby `Time` methods because specs that use timecop in
614
+ # before(:all) hooks will mess up our times.
615
+ def current_time
616
+ @redis.time[0]
617
+ end
618
+
619
+ # Reconstructs whole-file timings by summing the timings of a file's
620
+ # individual per-example ("file[...]") entries.
621
+ def populate_splitted_file_timings(timings)
622
+ whole_file_timings = Hash.new(0)
623
+
624
+ timings.each do |file, duration|
625
+ next if !file.include?("[")
626
+
627
+ base_file = file.split("[").first
628
+ whole_file_timings[base_file] += duration
629
+ end
630
+
631
+ whole_file_timings
632
+ end
633
+ end
634
+ end