rspecq-instructure 0.8.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/CHANGELOG.md +106 -0
- data/LICENSE +20 -0
- data/README.md +268 -0
- data/Rakefile +10 -0
- data/bin/rspecq +49 -0
- data/lib/rspecq/configuration.rb +103 -0
- data/lib/rspecq/formatters/README.md +4 -0
- data/lib/rspecq/formatters/example_count_recorder.rb +15 -0
- data/lib/rspecq/formatters/failure_recorder.rb +62 -0
- data/lib/rspecq/formatters/job_timing_recorder.rb +23 -0
- data/lib/rspecq/formatters/junit_formatter.rb +53 -0
- data/lib/rspecq/formatters/worker_heartbeat_recorder.rb +16 -0
- data/lib/rspecq/parser.rb +252 -0
- data/lib/rspecq/queue.rb +634 -0
- data/lib/rspecq/reporter.rb +195 -0
- data/lib/rspecq/version.rb +3 -0
- data/lib/rspecq/worker.rb +418 -0
- data/lib/rspecq.rb +15 -0
- metadata +206 -0
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
module RSpecQ
|
|
2
|
+
# A Reporter, given a build ID, is responsible for consolidating the results
|
|
3
|
+
# from different workers and printing a complete build summary to the user,
|
|
4
|
+
# along with any failures that might have occurred.
|
|
5
|
+
#
|
|
6
|
+
# The failures are printed in real-time as they occur, while the final
|
|
7
|
+
# summary is printed after the queue is empty and no tests are being
|
|
8
|
+
# executed. If the build failed, the status code of the reporter is non-zero.
|
|
9
|
+
#
|
|
10
|
+
# Reporters are readers of the queue.
|
|
11
|
+
class Reporter
|
|
12
|
+
def initialize(build_id:, timeout:, redis_opts:, worker_liveness_sec:, queue_wait_timeout: 30,
|
|
13
|
+
update_timings: false, timings_key: nil)
|
|
14
|
+
@build_id = build_id
|
|
15
|
+
@timeout = timeout
|
|
16
|
+
@queue = Queue.new(build_id, "reporter", redis_opts, worker_liveness_sec)
|
|
17
|
+
@queue_wait_timeout = queue_wait_timeout
|
|
18
|
+
@update_timings = update_timings
|
|
19
|
+
@timings_key = timings_key
|
|
20
|
+
|
|
21
|
+
# We want feedback to be immediately printed to CI users, so
|
|
22
|
+
# we disable buffering.
|
|
23
|
+
$stdout.sync = true
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
def report
|
|
27
|
+
@queue.wait_until_published(@queue_wait_timeout)
|
|
28
|
+
|
|
29
|
+
finished = false
|
|
30
|
+
|
|
31
|
+
reported_failures = {}
|
|
32
|
+
failure_heading_printed = false
|
|
33
|
+
|
|
34
|
+
@timeout.times do
|
|
35
|
+
@queue.example_failures.each do |job, rspec_output|
|
|
36
|
+
next if reported_failures[job]
|
|
37
|
+
|
|
38
|
+
if !failure_heading_printed
|
|
39
|
+
puts "\nFailures:\n"
|
|
40
|
+
failure_heading_printed = true
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
reported_failures[job] = true
|
|
44
|
+
puts failure_formatted(rspec_output)
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
unless @queue.exhausted? || @queue.build_failed_fast?
|
|
48
|
+
sleep 1
|
|
49
|
+
next
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
finished = true
|
|
53
|
+
break
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
raise "Build not finished after #{@timeout} seconds" if !finished
|
|
57
|
+
|
|
58
|
+
# The reporter can observe the build finished before any worker stamps
|
|
59
|
+
# finished_at; stamp it here (setnx, first writer wins) so the build
|
|
60
|
+
# duration — and our canvas-consumed key_build_time — are always recorded.
|
|
61
|
+
@queue.try_mark_finished
|
|
62
|
+
|
|
63
|
+
build_duration = test_durations&.first
|
|
64
|
+
@queue.record_build_time(build_duration) if build_duration
|
|
65
|
+
|
|
66
|
+
if @update_timings && @queue.build_successful?
|
|
67
|
+
if @timings_key
|
|
68
|
+
puts "Updating job timings @ #{@timings_key}"
|
|
69
|
+
@queue.update_global_timings(@timings_key)
|
|
70
|
+
else
|
|
71
|
+
puts "Updating global job timings"
|
|
72
|
+
@queue.update_global_timings
|
|
73
|
+
end
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
flaky_jobs = @queue.flaky_jobs
|
|
77
|
+
|
|
78
|
+
puts summary(@queue.example_failures, @queue.non_example_errors, flaky_jobs)
|
|
79
|
+
|
|
80
|
+
flaky_jobs_to_sentry(flaky_jobs, build_duration)
|
|
81
|
+
|
|
82
|
+
exit 1 if !@queue.build_successful?
|
|
83
|
+
end
|
|
84
|
+
|
|
85
|
+
private
|
|
86
|
+
|
|
87
|
+
# Two build durations (secs): from master election, and from queue ready.
|
|
88
|
+
# nil until the build has both a start and a finish timestamp.
|
|
89
|
+
def test_durations
|
|
90
|
+
@test_durations ||= @queue.took_times_secs
|
|
91
|
+
end
|
|
92
|
+
|
|
93
|
+
# We try to keep this output consistent with RSpec's original output
|
|
94
|
+
def summary(failures, errors, flaky_jobs)
|
|
95
|
+
failed_examples_section = "\nFailed examples:\n\n"
|
|
96
|
+
|
|
97
|
+
failures.each_value do |msg|
|
|
98
|
+
parts = msg.split("\n")
|
|
99
|
+
failed_examples_section << " #{parts[-1]}\n"
|
|
100
|
+
end
|
|
101
|
+
|
|
102
|
+
summary = ""
|
|
103
|
+
if @queue.build_failed_fast?
|
|
104
|
+
summary << "\n\n"
|
|
105
|
+
summary << "The limit of #{@queue.fail_fast} failures has been reached\n"
|
|
106
|
+
summary << "Aborting..."
|
|
107
|
+
summary << "\n"
|
|
108
|
+
end
|
|
109
|
+
|
|
110
|
+
summary << failed_examples_section if !failures.empty?
|
|
111
|
+
|
|
112
|
+
errors.each_value { |msg| summary << msg }
|
|
113
|
+
|
|
114
|
+
requeues = @queue.requeued_jobs.values.sum
|
|
115
|
+
|
|
116
|
+
summary << "\n"
|
|
117
|
+
summary << "Total results:\n"
|
|
118
|
+
summary << " #{@queue.example_count} examples " \
|
|
119
|
+
"(#{@queue.processed_jobs_count} jobs), " \
|
|
120
|
+
"#{failures.count} failures, " \
|
|
121
|
+
"#{errors.count} errors, " \
|
|
122
|
+
"#{requeues} requeues"
|
|
123
|
+
summary << ", #{flaky_jobs.count} flaky" if flaky_jobs.any?
|
|
124
|
+
summary << ", #{@queue.lost_jobs_count} lost jobs (unique)" if @queue.lost_jobs_count.positive?
|
|
125
|
+
summary << "\n\n\n"
|
|
126
|
+
|
|
127
|
+
from_elected_master, from_queue_ready = test_durations
|
|
128
|
+
if from_elected_master
|
|
129
|
+
summary << "Spec time (from elected master): #{humanize_duration(from_elected_master)}\n"
|
|
130
|
+
end
|
|
131
|
+
if from_queue_ready
|
|
132
|
+
summary << "Spec time (from queue ready): #{humanize_duration(from_queue_ready)}\n"
|
|
133
|
+
end
|
|
134
|
+
summary << "Worker total execution time: " \
|
|
135
|
+
"#{humanize_duration(@queue.total_execution_time_ms / 1000)}"
|
|
136
|
+
|
|
137
|
+
if !flaky_jobs.empty?
|
|
138
|
+
summary << "\n\n"
|
|
139
|
+
summary << "Flaky jobs detected (count=#{flaky_jobs.count}):\n"
|
|
140
|
+
flaky_jobs.each do |j|
|
|
141
|
+
job_timing = (jt = @queue.job_build_timing(j)) ? humanize_duration(jt.to_i) : "---"
|
|
142
|
+
summary << RSpec::Core::Formatters::ConsoleCodes.wrap(
|
|
143
|
+
"#{@queue.job_location(j)} @ #{@queue.failed_job_worker(j)} timing=#{job_timing}\n",
|
|
144
|
+
RSpec.configuration.pending_color
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
next if ENV["RSPECQ_REPORTER_RERUN_COMMAND_SKIP"]
|
|
148
|
+
|
|
149
|
+
summary << "#{@queue.job_rerun_command(j)}\n\n\n"
|
|
150
|
+
end
|
|
151
|
+
end
|
|
152
|
+
|
|
153
|
+
summary
|
|
154
|
+
end
|
|
155
|
+
|
|
156
|
+
def failure_formatted(rspec_output)
|
|
157
|
+
rspec_output.split("\n")[0..-2].join("\n")
|
|
158
|
+
end
|
|
159
|
+
|
|
160
|
+
def humanize_duration(secs)
|
|
161
|
+
min, sec = secs.divmod(60)
|
|
162
|
+
|
|
163
|
+
format("%<min>d:%<sec>02d", min: min, sec: sec)
|
|
164
|
+
end
|
|
165
|
+
|
|
166
|
+
def flaky_jobs_to_sentry(jobs, build_duration)
|
|
167
|
+
return if jobs.empty?
|
|
168
|
+
|
|
169
|
+
jobs.each do |job|
|
|
170
|
+
filename = job.gsub(%r{\[.+\]|\./}, "").split(":")[0]
|
|
171
|
+
|
|
172
|
+
extra = {
|
|
173
|
+
build: @build_id,
|
|
174
|
+
build_timeout: @timeout,
|
|
175
|
+
build_duration: build_duration,
|
|
176
|
+
location: @queue.job_location(job),
|
|
177
|
+
rerun_command: @queue.job_rerun_command(job),
|
|
178
|
+
worker: @queue.failed_job_worker(job)
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
tags = {
|
|
182
|
+
flaky: true,
|
|
183
|
+
spec_file: filename
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
Sentry.capture_message(
|
|
187
|
+
"Flaky test in #{filename}",
|
|
188
|
+
level: "warning",
|
|
189
|
+
extra: extra,
|
|
190
|
+
tags: tags
|
|
191
|
+
)
|
|
192
|
+
end
|
|
193
|
+
end
|
|
194
|
+
end
|
|
195
|
+
end
|
|
@@ -0,0 +1,418 @@
|
|
|
1
|
+
require "json"
|
|
2
|
+
require "pathname"
|
|
3
|
+
require "pp"
|
|
4
|
+
require "open3"
|
|
5
|
+
|
|
6
|
+
module RSpecQ
|
|
7
|
+
# A Worker, given a build ID, continuously consumes tests off the
|
|
8
|
+
# corresponding and executes them, until the queue is empty.
|
|
9
|
+
# It is also responsible for populating the initial queue.
|
|
10
|
+
#
|
|
11
|
+
# Essentially, a worker is an RSpec runner that prints the results of the
|
|
12
|
+
# tests it executes to standard output.
|
|
13
|
+
#
|
|
14
|
+
# The typical use case is to spawn many workers for a given build, thereby
|
|
15
|
+
# parallelizing the work and achieving faster build times.
|
|
16
|
+
#
|
|
17
|
+
# Workers are readers+writers of the queue.
|
|
18
|
+
class Worker
|
|
19
|
+
# The root path or individual spec files to execute.
|
|
20
|
+
#
|
|
21
|
+
# Defaults to "spec" (similar to RSpec)
|
|
22
|
+
attr_accessor :files_or_dirs_to_run
|
|
23
|
+
|
|
24
|
+
# If set, spec files that are known to take more than this value to finish,
|
|
25
|
+
# will be split and scheduled on a per-example basis.
|
|
26
|
+
#
|
|
27
|
+
# Defaults to 999999
|
|
28
|
+
attr_accessor :file_split_threshold
|
|
29
|
+
|
|
30
|
+
# Retry failed examples up to N times (with N being the supplied value)
|
|
31
|
+
# before considering them legit failures
|
|
32
|
+
#
|
|
33
|
+
# Defaults to 3
|
|
34
|
+
attr_accessor :max_requeues
|
|
35
|
+
|
|
36
|
+
# Stop the execution after N failed tests. Do not stop at any point when
|
|
37
|
+
# set to 0.
|
|
38
|
+
#
|
|
39
|
+
# Defaults to 0
|
|
40
|
+
attr_accessor :fail_fast
|
|
41
|
+
|
|
42
|
+
# Time to wait for a queue to be published.
|
|
43
|
+
#
|
|
44
|
+
# Defaults to 30
|
|
45
|
+
attr_accessor :queue_wait_timeout
|
|
46
|
+
|
|
47
|
+
# The RSpec seed
|
|
48
|
+
attr_accessor :seed
|
|
49
|
+
|
|
50
|
+
# Reproduction flag. If true, worker will publish files in the exact order
|
|
51
|
+
# given in the command.
|
|
52
|
+
attr_accessor :reproduction
|
|
53
|
+
|
|
54
|
+
# Include a suite counter in any output filenames so that each suite run
|
|
55
|
+
# Output Junit formatted XML
|
|
56
|
+
# Output Junit formatted XML to a specified file
|
|
57
|
+
#
|
|
58
|
+
# Example: test_results/results-{{TEST_ENV_NUMBER}}-{{JOB_INDEX}}.xml
|
|
59
|
+
# where TEST_ENV_NUMBER is substituted with the environment variable
|
|
60
|
+
# from the gem parallel test, and JOB_INDEX is incremented based
|
|
61
|
+
# on the number of test suites run in the current process.
|
|
62
|
+
attr_accessor :junit_output
|
|
63
|
+
|
|
64
|
+
# Optional arguments to pass along to rspec.
|
|
65
|
+
#
|
|
66
|
+
# Defaults to nil
|
|
67
|
+
attr_accessor :rspec_args
|
|
68
|
+
|
|
69
|
+
# RSpec tags to filter examples by (e.g. "slow" or "~slow"). Repeatable.
|
|
70
|
+
#
|
|
71
|
+
# Defaults to []
|
|
72
|
+
attr_accessor :tags
|
|
73
|
+
|
|
74
|
+
# Target duration in seconds for time-balanced example chunks.
|
|
75
|
+
# When splitting slow files, examples are grouped into chunks of
|
|
76
|
+
# approximately this duration to reduce Kernel.load calls.
|
|
77
|
+
#
|
|
78
|
+
# Defaults to 30
|
|
79
|
+
attr_accessor :chunk_target_duration
|
|
80
|
+
|
|
81
|
+
attr_reader :queue, :build_id, :worker_id
|
|
82
|
+
|
|
83
|
+
def initialize(build_id:, worker_id:, redis_opts:, worker_liveness_sec:)
|
|
84
|
+
@build_id = build_id
|
|
85
|
+
@worker_id = worker_id
|
|
86
|
+
@worker_liveness_sec = worker_liveness_sec
|
|
87
|
+
@heartbeat_frequency = worker_liveness_sec / 6
|
|
88
|
+
@queue = Queue.new(build_id, worker_id, redis_opts, worker_liveness_sec)
|
|
89
|
+
@fail_fast = 0
|
|
90
|
+
@files_or_dirs_to_run = ["spec"]
|
|
91
|
+
@file_split_threshold = 999_999
|
|
92
|
+
@heartbeat_updated_at = nil
|
|
93
|
+
@max_requeues = 3
|
|
94
|
+
@queue_wait_timeout = 30
|
|
95
|
+
@seed = srand && (srand % 0xFFFF)
|
|
96
|
+
@reproduction = false
|
|
97
|
+
@tags = []
|
|
98
|
+
@junit_output = nil
|
|
99
|
+
@chunk_target_duration = 30
|
|
100
|
+
|
|
101
|
+
RSpec::Core::Formatters.register(Formatters::JobTimingRecorder, :dump_summary)
|
|
102
|
+
RSpec::Core::Formatters.register(Formatters::ExampleCountRecorder, :dump_summary)
|
|
103
|
+
RSpec::Core::Formatters.register(Formatters::FailureRecorder, :example_failed, :message)
|
|
104
|
+
RSpec::Core::Formatters.register(Formatters::WorkerHeartbeatRecorder, :example_finished)
|
|
105
|
+
RSpec::Core::Formatters.register(Formatters::JUnitFormatter, :example_passed, :example_failed,
|
|
106
|
+
:start, :stop, :dump_summary)
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
def work
|
|
110
|
+
puts "Working for build #{@build_id} (worker=#{@worker_id})"
|
|
111
|
+
|
|
112
|
+
# If this worker crashed previously (OOM kill, etc.), it may have a job
|
|
113
|
+
# stuck in the running hash. Recover it before doing anything else —
|
|
114
|
+
# otherwise reserve_job will silently overwrite it.
|
|
115
|
+
recovered = queue.recover_own_job
|
|
116
|
+
puts "Recovered abandoned job from previous crash: #{recovered}" if recovered
|
|
117
|
+
|
|
118
|
+
q_size = try_publish_queue!(queue)
|
|
119
|
+
puts "Published queue (size=#{q_size})" if q_size
|
|
120
|
+
queue.wait_until_published(queue_wait_timeout)
|
|
121
|
+
queue.save_worker_seed(@worker_id, seed)
|
|
122
|
+
|
|
123
|
+
# Use `--seed` to deterministically reproduce test failures
|
|
124
|
+
# related to randomization by passing the same `--seed` value
|
|
125
|
+
# as the one that triggered the failure.
|
|
126
|
+
#
|
|
127
|
+
# We also use the same seed to feed Rspec's `--seed` option.
|
|
128
|
+
Kernel.srand(seed)
|
|
129
|
+
|
|
130
|
+
idx = 0
|
|
131
|
+
loop do
|
|
132
|
+
# we have to bootstrap this so that it can be used in the first call
|
|
133
|
+
# to `requeue_lost_job` inside the work loop
|
|
134
|
+
update_heartbeat
|
|
135
|
+
|
|
136
|
+
if queue.build_failed_fast?
|
|
137
|
+
queue.try_mark_finished
|
|
138
|
+
return
|
|
139
|
+
end
|
|
140
|
+
|
|
141
|
+
lost, lost_worker = queue.requeue_lost_job
|
|
142
|
+
puts "Requeued lost job: #{lost} from #{lost_worker}" if lost
|
|
143
|
+
|
|
144
|
+
# TODO: can we make `reserve_job` also act like exhausted? and get
|
|
145
|
+
# rid of `exhausted?` (i.e. return false if no jobs remain)
|
|
146
|
+
job = queue.reserve_job
|
|
147
|
+
|
|
148
|
+
# build is finished
|
|
149
|
+
if job.nil? && queue.exhausted?
|
|
150
|
+
queue.try_mark_finished
|
|
151
|
+
return
|
|
152
|
+
end
|
|
153
|
+
|
|
154
|
+
if job.nil?
|
|
155
|
+
# backoff if no job is available
|
|
156
|
+
sleep 1
|
|
157
|
+
next
|
|
158
|
+
end
|
|
159
|
+
|
|
160
|
+
puts
|
|
161
|
+
puts "Executing #{job}"
|
|
162
|
+
|
|
163
|
+
ENV["ERROR_CONTEXT_BASE_PATH"] = nil
|
|
164
|
+
unless queue.is_requeue(job).nil?
|
|
165
|
+
ENV["ERROR_CONTEXT_BASE_PATH"] = "/usr/src/app/log/spec_failures/Rerun_#{queue.is_requeue(job)}"
|
|
166
|
+
end
|
|
167
|
+
|
|
168
|
+
reset_rspec_state!
|
|
169
|
+
|
|
170
|
+
# reconfigure rspec
|
|
171
|
+
RSpec.configuration.detail_color = :magenta
|
|
172
|
+
RSpec.configuration.seed = seed
|
|
173
|
+
RSpec.configuration.backtrace_formatter.filter_gem("rspecq")
|
|
174
|
+
RSpec.configuration.add_formatter(Formatters::FailureRecorder.new(queue, job, max_requeues, @worker_id))
|
|
175
|
+
|
|
176
|
+
if junit_output
|
|
177
|
+
RSpec.configuration.add_formatter(Formatters::JUnitFormatter.new(queue, job, max_requeues,
|
|
178
|
+
idx, junit_output))
|
|
179
|
+
end
|
|
180
|
+
|
|
181
|
+
RSpec.configuration.add_formatter(Formatters::ExampleCountRecorder.new(queue))
|
|
182
|
+
RSpec.configuration.add_formatter(Formatters::WorkerHeartbeatRecorder.new(self))
|
|
183
|
+
|
|
184
|
+
# Recording is always-on: every build records per-job timings into the
|
|
185
|
+
# build-scoped key. The reporter promotes them to the global key only
|
|
186
|
+
# when --update-timings is set.
|
|
187
|
+
RSpec.configuration.add_formatter(Formatters::JobTimingRecorder.new(queue, job))
|
|
188
|
+
|
|
189
|
+
args = [*rspec_args, "--format", "progress", *job.split("+")]
|
|
190
|
+
tags.each { |tag| args.push("--tag", tag) }
|
|
191
|
+
opts = RSpec::Core::ConfigurationOptions.new(args)
|
|
192
|
+
|
|
193
|
+
_result = RSpec::Core::Runner.new(opts).run($stderr, $stdout)
|
|
194
|
+
|
|
195
|
+
queue.acknowledge_job(job)
|
|
196
|
+
idx += 1
|
|
197
|
+
end
|
|
198
|
+
end
|
|
199
|
+
|
|
200
|
+
# Update the worker heartbeat if necessary
|
|
201
|
+
def update_heartbeat
|
|
202
|
+
if @heartbeat_updated_at.nil? || elapsed(@heartbeat_updated_at) >= @heartbeat_frequency
|
|
203
|
+
queue.record_worker_heartbeat
|
|
204
|
+
@heartbeat_updated_at = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
205
|
+
end
|
|
206
|
+
end
|
|
207
|
+
|
|
208
|
+
# Memoized global timings — read once and reused across the scheduling path.
|
|
209
|
+
def global_timings
|
|
210
|
+
@global_timings ||= queue.global_timings
|
|
211
|
+
end
|
|
212
|
+
|
|
213
|
+
def try_publish_queue!(queue)
|
|
214
|
+
return if !queue.become_master
|
|
215
|
+
|
|
216
|
+
queue.mark_elected_master_at
|
|
217
|
+
|
|
218
|
+
if reproduction
|
|
219
|
+
q_size = queue.publish(files_or_dirs_to_run, fail_fast)
|
|
220
|
+
log_event(
|
|
221
|
+
"Reproduction mode. Published queue as given (size=#{q_size})",
|
|
222
|
+
"info"
|
|
223
|
+
)
|
|
224
|
+
return q_size
|
|
225
|
+
end
|
|
226
|
+
|
|
227
|
+
puts "I am the master worker (worker_id=#{@worker_id}), publishing the queue..."
|
|
228
|
+
|
|
229
|
+
RSpec.configuration.files_or_directories_to_run = files_or_dirs_to_run
|
|
230
|
+
files_to_run = RSpec.configuration.files_to_run.map { |j| relative_path(j) }
|
|
231
|
+
|
|
232
|
+
if global_timings.empty?
|
|
233
|
+
q_size = queue.publish(files_to_run.shuffle, fail_fast)
|
|
234
|
+
log_event(
|
|
235
|
+
"No timings found! Published queue in random order (size=#{q_size})",
|
|
236
|
+
"warning"
|
|
237
|
+
)
|
|
238
|
+
return q_size
|
|
239
|
+
end
|
|
240
|
+
|
|
241
|
+
# prepare jobs to run
|
|
242
|
+
jobs = []
|
|
243
|
+
slow_files = []
|
|
244
|
+
|
|
245
|
+
if file_split_threshold
|
|
246
|
+
slow_files = global_timings.take_while do |_job, duration|
|
|
247
|
+
duration >= file_split_threshold
|
|
248
|
+
end.map(&:first) & files_to_run
|
|
249
|
+
end
|
|
250
|
+
|
|
251
|
+
if slow_files.any?
|
|
252
|
+
jobs.concat(files_to_run - slow_files)
|
|
253
|
+
example_ids = files_to_example_ids(slow_files)
|
|
254
|
+
chunks = build_time_balanced_chunks(example_ids, global_timings, chunk_target_duration)
|
|
255
|
+
jobs.concat(chunks)
|
|
256
|
+
else
|
|
257
|
+
jobs.concat(files_to_run)
|
|
258
|
+
end
|
|
259
|
+
|
|
260
|
+
jobs = order_jobs_by_timings(jobs)
|
|
261
|
+
|
|
262
|
+
queue.publish(jobs, fail_fast)
|
|
263
|
+
end
|
|
264
|
+
|
|
265
|
+
private
|
|
266
|
+
|
|
267
|
+
# Assign each job its previous timing (chunk jobs sum their examples;
|
|
268
|
+
# untimed jobs get the median so they land mid-queue), then order slowest
|
|
269
|
+
# first so the longest jobs start earliest.
|
|
270
|
+
def order_jobs_by_timings(jobs)
|
|
271
|
+
default_timing = global_timings.values[global_timings.values.size / 2]
|
|
272
|
+
|
|
273
|
+
jobs = jobs.each_with_object({}) do |j, h|
|
|
274
|
+
if j.include?("+")
|
|
275
|
+
parts = j.split("+")
|
|
276
|
+
h[j] = parts.sum { |p| global_timings[p] || default_timing }
|
|
277
|
+
else
|
|
278
|
+
puts "Untimed job: #{j}" if global_timings[j].nil?
|
|
279
|
+
h[j] = global_timings[j] || default_timing
|
|
280
|
+
end
|
|
281
|
+
end
|
|
282
|
+
|
|
283
|
+
jobs.sort_by { |_j, t| -t }.map(&:first)
|
|
284
|
+
end
|
|
285
|
+
|
|
286
|
+
# Groups example IDs into time-balanced chunks, one chunk per Kernel.load.
|
|
287
|
+
# Examples from different files are never mixed. Uses per-example timings
|
|
288
|
+
# from Redis when available; falls back to file_timing / example_count.
|
|
289
|
+
#
|
|
290
|
+
# Returns an array of job strings: single examples are bare IDs, multi-example
|
|
291
|
+
# chunks are "+"-delimited (e.g. "spec/foo.rb[1:1]+spec/foo.rb[1:2]").
|
|
292
|
+
def build_time_balanced_chunks(example_ids, timings, target_duration)
|
|
293
|
+
by_file = example_ids.group_by { |id| id.sub(/\[.*\]$/, "") }
|
|
294
|
+
chunks = []
|
|
295
|
+
|
|
296
|
+
by_file.each do |file, ids|
|
|
297
|
+
file_timing = timings[file]
|
|
298
|
+
default_example_timing = file_timing ? file_timing / ids.size : target_duration / 10.0
|
|
299
|
+
|
|
300
|
+
timed_ids = ids.map { |id| [id, timings[id] || default_example_timing] }
|
|
301
|
+
timed_ids.sort_by! { |_, t| -t }
|
|
302
|
+
|
|
303
|
+
current_chunk = []
|
|
304
|
+
current_duration = 0.0
|
|
305
|
+
|
|
306
|
+
timed_ids.each do |id, duration|
|
|
307
|
+
if current_chunk.empty? || current_duration + duration <= target_duration
|
|
308
|
+
current_chunk << id
|
|
309
|
+
current_duration += duration
|
|
310
|
+
else
|
|
311
|
+
chunks << current_chunk.join("+")
|
|
312
|
+
current_chunk = [id]
|
|
313
|
+
current_duration = duration
|
|
314
|
+
end
|
|
315
|
+
end
|
|
316
|
+
chunks << current_chunk.join("+") unless current_chunk.empty?
|
|
317
|
+
end
|
|
318
|
+
|
|
319
|
+
chunks
|
|
320
|
+
end
|
|
321
|
+
|
|
322
|
+
def reset_rspec_state!
|
|
323
|
+
RSpec.clear_examples
|
|
324
|
+
|
|
325
|
+
# see https://github.com/rspec/rspec-core/pull/2723
|
|
326
|
+
if Gem::Version.new(RSpec::Core::Version::STRING) <= Gem::Version.new("3.9.1")
|
|
327
|
+
RSpec.world.instance_variable_set(
|
|
328
|
+
:@example_group_counts_by_spec_file, Hash.new(0)
|
|
329
|
+
)
|
|
330
|
+
end
|
|
331
|
+
|
|
332
|
+
# RSpec.clear_examples does not reset those, which causes issues when
|
|
333
|
+
# a non-example error occurs (subsequent jobs are not executed)
|
|
334
|
+
# TODO: upstream
|
|
335
|
+
RSpec.world.non_example_failure = false
|
|
336
|
+
|
|
337
|
+
# we don't want an error that occured outside of the examples (which
|
|
338
|
+
# would set this to `true`) to stop the worker
|
|
339
|
+
RSpec.world.wants_to_quit = false
|
|
340
|
+
|
|
341
|
+
# RSpec.clear_examples calls world.reset which clears world.example_groups,
|
|
342
|
+
# but NOT world.filtered_examples. That hash is keyed by example group
|
|
343
|
+
# class objects, so old group classes from the previous job remain
|
|
344
|
+
# referenced as hash keys and cannot be GC'd.
|
|
345
|
+
RSpec.world.filtered_examples.clear
|
|
346
|
+
|
|
347
|
+
# The shared example group registry is also never cleared by world.reset.
|
|
348
|
+
# When a spec file defines shared_examples/shared_context inside a
|
|
349
|
+
# describe block, the example group CLASS becomes a key in the registry's
|
|
350
|
+
# internal hash. Each `load` of a spec file creates new group classes
|
|
351
|
+
# that get pinned as hash keys, preventing GC of the entire class tree
|
|
352
|
+
# (including onceler Marshal blobs). We clear per-group entries but
|
|
353
|
+
# preserve top-level (:main) shared examples from support files, which
|
|
354
|
+
# are loaded once via `require` and won't be re-defined.
|
|
355
|
+
RSpec.world.shared_example_group_registry
|
|
356
|
+
.send(:shared_example_groups)
|
|
357
|
+
.select! { |k, _| k == :main }
|
|
358
|
+
end
|
|
359
|
+
|
|
360
|
+
# NOTE: RSpec has to load the files before we can split them as individual
|
|
361
|
+
# examples. In case a file to be splitted fails to be loaded
|
|
362
|
+
# (e.g. contains a syntax error), we return the files unchanged, thereby
|
|
363
|
+
# falling back to scheduling them as whole files. Their errors will be
|
|
364
|
+
# reported in the normal flow when they're eventually picked up by a worker.
|
|
365
|
+
def files_to_example_ids(files)
|
|
366
|
+
cmd = "DISABLE_SPRING=1 SUPPRESS_OUTPUT=1 bundle exec rspec --dry-run --format json #{files.join(' ')}"
|
|
367
|
+
out, err, cmd_result = Open3.capture3(cmd)
|
|
368
|
+
|
|
369
|
+
if !cmd_result.success?
|
|
370
|
+
rspec_output = begin
|
|
371
|
+
JSON.parse(out)
|
|
372
|
+
rescue JSON::ParserError
|
|
373
|
+
out
|
|
374
|
+
end
|
|
375
|
+
|
|
376
|
+
log_event(
|
|
377
|
+
"Failed to split slow files, falling back to regular scheduling.\n #{err}",
|
|
378
|
+
"error",
|
|
379
|
+
rspec_stdout: rspec_output,
|
|
380
|
+
rspec_stderr: err,
|
|
381
|
+
cmd_result: cmd_result.inspect
|
|
382
|
+
)
|
|
383
|
+
|
|
384
|
+
pp rspec_output
|
|
385
|
+
|
|
386
|
+
return files
|
|
387
|
+
end
|
|
388
|
+
|
|
389
|
+
JSON.parse(out)["examples"].map { |e| e["id"] }
|
|
390
|
+
end
|
|
391
|
+
|
|
392
|
+
def relative_path(job)
|
|
393
|
+
@cwd ||= Pathname.new(Dir.pwd)
|
|
394
|
+
"./#{Pathname.new(job).relative_path_from(@cwd)}"
|
|
395
|
+
end
|
|
396
|
+
|
|
397
|
+
def elapsed(since)
|
|
398
|
+
Process.clock_gettime(Process::CLOCK_MONOTONIC) - since
|
|
399
|
+
end
|
|
400
|
+
|
|
401
|
+
# Prints msg to standard output and emits an event to Sentry, if the
|
|
402
|
+
# SENTRY_DSN environment variable is set.
|
|
403
|
+
def log_event(msg, level, additional = {})
|
|
404
|
+
puts msg
|
|
405
|
+
|
|
406
|
+
Sentry.capture_message(msg, level: level, extra: {
|
|
407
|
+
build: @build_id,
|
|
408
|
+
worker: @worker_id,
|
|
409
|
+
queue: queue.inspect,
|
|
410
|
+
files_or_dirs_to_run: files_or_dirs_to_run,
|
|
411
|
+
file_split_threshold: file_split_threshold,
|
|
412
|
+
heartbeat_updated_at: @heartbeat_updated_at,
|
|
413
|
+
object: inspect,
|
|
414
|
+
pid: Process.pid
|
|
415
|
+
}.merge(additional))
|
|
416
|
+
end
|
|
417
|
+
end
|
|
418
|
+
end
|
data/lib/rspecq.rb
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
require "rspec/core"
|
|
2
|
+
require "sentry-ruby"
|
|
3
|
+
|
|
4
|
+
require_relative "rspecq/formatters/example_count_recorder"
|
|
5
|
+
require_relative "rspecq/formatters/failure_recorder"
|
|
6
|
+
require_relative "rspecq/formatters/job_timing_recorder"
|
|
7
|
+
require_relative "rspecq/formatters/junit_formatter"
|
|
8
|
+
require_relative "rspecq/formatters/worker_heartbeat_recorder"
|
|
9
|
+
|
|
10
|
+
require_relative "rspecq/configuration"
|
|
11
|
+
require_relative "rspecq/parser"
|
|
12
|
+
require_relative "rspecq/queue"
|
|
13
|
+
require_relative "rspecq/reporter"
|
|
14
|
+
require_relative "rspecq/version"
|
|
15
|
+
require_relative "rspecq/worker"
|