monkrb 0.17.0 → 0.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +56 -0
- data/README.md +5 -4
- data/exe/monk +45 -7
- data/lib/monk/jobs/adapters/pg.rb +272 -0
- data/lib/monk/jobs/args.rb +43 -0
- data/lib/monk/jobs/claim.rb +7 -0
- data/lib/monk/jobs/errors.rb +29 -0
- data/lib/monk/jobs/job.rb +117 -0
- data/lib/monk/jobs/runtime.rb +239 -0
- data/lib/monk/jobs/worker.rb +94 -0
- data/lib/monk/jobs.rb +229 -0
- data/lib/monk/mail/errors.rb +7 -0
- data/lib/monk/mail/later.rb +53 -0
- data/lib/monk/persistence/pg.rb +21 -1
- data/lib/monk/scaffold.rb +232 -8
- data/lib/monk/templates/jobs/bin/jobs +26 -0
- data/lib/monk/templates/jobs/config/jobs.rb +17 -0
- data/lib/monk/templates/jobs/db/migrate/00000000000002_create_jobs_tables.down.sql +3 -0
- data/lib/monk/templates/jobs/db/migrate/00000000000002_create_jobs_tables.up.sql +48 -0
- data/lib/monk/templates/jobs/jobs/hello_job.rb +9 -0
- data/lib/monk/templates/jobs/jobs/send_login_link.rb +27 -0
- data/lib/monk/templates/postgres/Dockerfile +3 -1
- data/lib/monk/version.rb +1 -1
- metadata +16 -1
|
@@ -0,0 +1,239 @@
|
|
|
1
|
+
require "socket"
|
|
2
|
+
require_relative "../jobs"
|
|
3
|
+
require_relative "worker"
|
|
4
|
+
|
|
5
|
+
module Monk
|
|
6
|
+
module Jobs
|
|
7
|
+
# A job process (bin/jobs): the supervisor in the calling Ractor and N
|
|
8
|
+
# worker Ractors (docs/history/plan-jobs.md Phase 6). Opt-in on its own,
|
|
9
|
+
# require "monk/jobs/runtime" -- the web process only enqueues.
|
|
10
|
+
#
|
|
11
|
+
# Monk::Jobs::Runtime.new(queues: %w[mailers default], workers: 5).run
|
|
12
|
+
#
|
|
13
|
+
# run blocks until TERM or INT (or #stop): workers finish the job in
|
|
14
|
+
# hand, up to shutdown_timeout seconds, and whatever is still running
|
|
15
|
+
# then is released for another process.
|
|
16
|
+
class Runtime
|
|
17
|
+
DEFAULTS = {
|
|
18
|
+
queues: [Job::DEFAULT_QUEUE], workers: 5, poll_interval: 1.0, tick_interval: 1.0,
|
|
19
|
+
heartbeat_interval: 15, process_timeout: 120, shutdown_timeout: 25,
|
|
20
|
+
}.freeze
|
|
21
|
+
|
|
22
|
+
# queues: claimed in this order, so earlier ones go first.
|
|
23
|
+
# poll_interval: seconds an idle worker waits before claiming again.
|
|
24
|
+
# tick_interval: seconds between the supervisor's rounds of staging
|
|
25
|
+
# due jobs (so also how late a scheduled job can start).
|
|
26
|
+
# heartbeat_interval: seconds between this process's heartbeats, and
|
|
27
|
+
# between its looks for dead processes to prune.
|
|
28
|
+
# process_timeout: seconds of silence after which another process's
|
|
29
|
+
# running jobs are released.
|
|
30
|
+
# shutdown_timeout: seconds a stop waits for jobs in flight.
|
|
31
|
+
def initialize(**settings)
|
|
32
|
+
unknown = settings.keys - DEFAULTS.keys
|
|
33
|
+
raise ArgumentError, "unknown Monk::Jobs::Runtime setting(s): #{unknown.join(", ")}" if unknown.any?
|
|
34
|
+
|
|
35
|
+
settings = DEFAULTS.merge(settings)
|
|
36
|
+
@queues = checked_queues(settings[:queues])
|
|
37
|
+
@worker_count = settings[:workers]
|
|
38
|
+
unless positive_integer?(@worker_count)
|
|
39
|
+
raise ArgumentError, "workers must be a positive Integer, got #{@worker_count.inspect}"
|
|
40
|
+
end
|
|
41
|
+
|
|
42
|
+
%i[poll_interval tick_interval heartbeat_interval process_timeout shutdown_timeout].each do |name|
|
|
43
|
+
value = settings[name]
|
|
44
|
+
unless value.is_a?(Numeric) && value.positive?
|
|
45
|
+
raise ArgumentError, "#{name} must be a positive number of seconds, got #{value.inspect}"
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
instance_variable_set(:"@#{name}", value)
|
|
49
|
+
end
|
|
50
|
+
@stopping = false
|
|
51
|
+
end
|
|
52
|
+
|
|
53
|
+
def run(trap_signals: true)
|
|
54
|
+
Monk.freeze!
|
|
55
|
+
@adapter = Monk::Jobs.adapter
|
|
56
|
+
@hostname = Socket.gethostname
|
|
57
|
+
@pid = Process.pid
|
|
58
|
+
@process_id = @adapter.register_process(hostname: @hostname, pid: @pid)
|
|
59
|
+
trap_signals! if trap_signals
|
|
60
|
+
|
|
61
|
+
@monitors = {} # monitor port => [worker index, its Ractor]
|
|
62
|
+
@controls = {} # worker index => its control port
|
|
63
|
+
@holding = {} # worker index => id of the job it's running
|
|
64
|
+
@to_release = [] # ids of jobs dead workers held, not yet released
|
|
65
|
+
@holding_port = Ractor::Port.new
|
|
66
|
+
@worker_count.times { |index| spawn_worker(index) }
|
|
67
|
+
Monk::Log.info(
|
|
68
|
+
"Monk::Jobs process #{@process_id} started: #{@worker_count} workers on #{@queues.join(", ")} (pid #{@pid})",
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
ticker, timer = start_ticker
|
|
72
|
+
supervise(ticker)
|
|
73
|
+
shut_down(ticker)
|
|
74
|
+
ensure
|
|
75
|
+
timer&.kill
|
|
76
|
+
ticker&.close
|
|
77
|
+
@holding_port&.close
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
# Asks run to stop; safe from a signal handler or another thread.
|
|
81
|
+
def stop
|
|
82
|
+
@stopping = true
|
|
83
|
+
end
|
|
84
|
+
|
|
85
|
+
private
|
|
86
|
+
|
|
87
|
+
def supervise(ticker)
|
|
88
|
+
last_heartbeat = monotonic
|
|
89
|
+
dead = []
|
|
90
|
+
until @stopping
|
|
91
|
+
port, message = Ractor.select(ticker, @holding_port, *@monitors.keys)
|
|
92
|
+
if port.equal?(ticker)
|
|
93
|
+
last_heartbeat = tick(last_heartbeat)
|
|
94
|
+
# Respawned on the next tick rather than at once, so a worker
|
|
95
|
+
# that dies straight away (the database is down) can't spin.
|
|
96
|
+
dead.each { |index| spawn_worker(index) }
|
|
97
|
+
dead.clear
|
|
98
|
+
elsif port.equal?(@holding_port)
|
|
99
|
+
index, id = message
|
|
100
|
+
id ? @holding[index] = id : @holding.delete(index)
|
|
101
|
+
else
|
|
102
|
+
dead << worker_exited(port)
|
|
103
|
+
end
|
|
104
|
+
end
|
|
105
|
+
end
|
|
106
|
+
|
|
107
|
+
# A database error here (Postgres restarting, a dropped connection)
|
|
108
|
+
# mustn't end the process: it's logged, this Ractor's connection is
|
|
109
|
+
# reset, and the next tick tries again.
|
|
110
|
+
def tick(last_heartbeat)
|
|
111
|
+
release_dead_workers_jobs
|
|
112
|
+
@adapter.stage_due
|
|
113
|
+
return last_heartbeat if monotonic - last_heartbeat < @heartbeat_interval
|
|
114
|
+
|
|
115
|
+
beat
|
|
116
|
+
monotonic
|
|
117
|
+
rescue StandardError => e
|
|
118
|
+
Monk::Log.error("Monk::Jobs supervisor: #{Monk::Jobs.describe_error(e)}; reconnecting")
|
|
119
|
+
reconnect
|
|
120
|
+
last_heartbeat
|
|
121
|
+
end
|
|
122
|
+
|
|
123
|
+
# A job stays on the list until its release has gone through (or found
|
|
124
|
+
# the job no longer held), so a database error just retries it next tick.
|
|
125
|
+
def release_dead_workers_jobs
|
|
126
|
+
until @to_release.empty?
|
|
127
|
+
@adapter.release(@to_release.first, @process_id)
|
|
128
|
+
@to_release.shift
|
|
129
|
+
end
|
|
130
|
+
end
|
|
131
|
+
|
|
132
|
+
def reconnect
|
|
133
|
+
@adapter.reset_connection
|
|
134
|
+
rescue StandardError => e
|
|
135
|
+
Monk::Log.error("Monk::Jobs supervisor: can't reconnect yet (#{e.class}: #{e.message})")
|
|
136
|
+
end
|
|
137
|
+
|
|
138
|
+
def beat
|
|
139
|
+
@adapter.heartbeat(@process_id, hostname: @hostname, pid: @pid)
|
|
140
|
+
released = @adapter.prune(@process_timeout)
|
|
141
|
+
return unless released.positive?
|
|
142
|
+
|
|
143
|
+
Monk::Log.warn("Monk::Jobs: released #{released} job(s) of processes silent for #{@process_timeout}s")
|
|
144
|
+
end
|
|
145
|
+
|
|
146
|
+
def spawn_worker(index)
|
|
147
|
+
hello = Ractor::Port.new
|
|
148
|
+
args = [@adapter, @process_id, @queues, @poll_interval, hello, @holding_port, index]
|
|
149
|
+
ractor = Ractor.new(*args, name: "monk-jobs-worker-#{index}") { |*worker_args| Monk::Jobs::Worker.run(*worker_args) }
|
|
150
|
+
@controls[index] = hello.receive
|
|
151
|
+
monitor = Ractor::Port.new
|
|
152
|
+
ractor.monitor(monitor)
|
|
153
|
+
@monitors[monitor] = [index, ractor]
|
|
154
|
+
ensure
|
|
155
|
+
hello&.close
|
|
156
|
+
end
|
|
157
|
+
|
|
158
|
+
# One port per worker: Ractor#monitor's message is a bare :exited or
|
|
159
|
+
# :aborted that doesn't say which Ractor (Phase 0.4).
|
|
160
|
+
def worker_exited(port)
|
|
161
|
+
index, ractor = @monitors.delete(port)
|
|
162
|
+
@controls.delete(index)
|
|
163
|
+
held = @holding.delete(index)
|
|
164
|
+
@to_release << held if held
|
|
165
|
+
begin
|
|
166
|
+
ractor.value
|
|
167
|
+
Monk::Log.warn("Monk::Jobs: worker #{index} exited; starting a new one")
|
|
168
|
+
rescue Ractor::RemoteError => e
|
|
169
|
+
Monk::Log.error("Monk::Jobs: worker #{index} died; starting a new one. #{Monk::Jobs.describe_error(e.cause)}")
|
|
170
|
+
end
|
|
171
|
+
index
|
|
172
|
+
end
|
|
173
|
+
|
|
174
|
+
def shut_down(ticker)
|
|
175
|
+
Monk::Log.info(
|
|
176
|
+
"Monk::Jobs process #{@process_id} stopping: waiting up to #{@shutdown_timeout}s for jobs in flight",
|
|
177
|
+
)
|
|
178
|
+
@controls.each_value do |control|
|
|
179
|
+
control << :stop
|
|
180
|
+
rescue Ractor::ClosedError
|
|
181
|
+
nil
|
|
182
|
+
end
|
|
183
|
+
|
|
184
|
+
deadline = monotonic + @shutdown_timeout
|
|
185
|
+
until @monitors.empty? || monotonic >= deadline
|
|
186
|
+
port, = Ractor.select(ticker, *@monitors.keys)
|
|
187
|
+
@monitors.delete(port) unless port.equal?(ticker)
|
|
188
|
+
end
|
|
189
|
+
unless @monitors.empty?
|
|
190
|
+
Monk::Log.warn(
|
|
191
|
+
"Monk::Jobs: #{@monitors.size} worker(s) still busy after #{@shutdown_timeout}s; releasing their jobs",
|
|
192
|
+
)
|
|
193
|
+
end
|
|
194
|
+
|
|
195
|
+
begin
|
|
196
|
+
@adapter.deregister(@process_id)
|
|
197
|
+
rescue StandardError => e
|
|
198
|
+
# Its jobs are released by another process's pruning instead.
|
|
199
|
+
Monk::Log.error("Monk::Jobs process #{@process_id} couldn't deregister: #{e.class}: #{e.message}")
|
|
200
|
+
end
|
|
201
|
+
Monk::Log.info("Monk::Jobs process #{@process_id} stopped")
|
|
202
|
+
end
|
|
203
|
+
|
|
204
|
+
def start_ticker
|
|
205
|
+
ticker = Ractor::Port.new
|
|
206
|
+
interval = @tick_interval
|
|
207
|
+
timer = Thread.new do
|
|
208
|
+
loop do
|
|
209
|
+
sleep interval
|
|
210
|
+
ticker << :tick
|
|
211
|
+
end
|
|
212
|
+
rescue Ractor::ClosedError
|
|
213
|
+
nil
|
|
214
|
+
end
|
|
215
|
+
[ticker, timer]
|
|
216
|
+
end
|
|
217
|
+
|
|
218
|
+
def trap_signals!
|
|
219
|
+
%w[TERM INT].each { |signal| Signal.trap(signal) { stop } }
|
|
220
|
+
end
|
|
221
|
+
|
|
222
|
+
def checked_queues(queues)
|
|
223
|
+
unless queues.is_a?(Array) && queues.any? && queues.all? { |q| q.is_a?(String) && !q.empty? }
|
|
224
|
+
raise ArgumentError, "queues must be a non-empty Array of queue names, got #{queues.inspect}"
|
|
225
|
+
end
|
|
226
|
+
|
|
227
|
+
Ractor.make_shareable(queues.map(&:dup))
|
|
228
|
+
end
|
|
229
|
+
|
|
230
|
+
def positive_integer?(value)
|
|
231
|
+
value.is_a?(Integer) && value.positive?
|
|
232
|
+
end
|
|
233
|
+
|
|
234
|
+
def monotonic
|
|
235
|
+
Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
236
|
+
end
|
|
237
|
+
end
|
|
238
|
+
end
|
|
239
|
+
end
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
require "timeout"
|
|
2
|
+
|
|
3
|
+
module Monk
|
|
4
|
+
module Jobs
|
|
5
|
+
# One worker Ractor's loop (docs/history/plan-jobs.md Phase 6): claim
|
|
6
|
+
# from the queues in order, run the job, finish or fail it, and wait
|
|
7
|
+
# poll_interval when there's nothing to claim. Plain module methods,
|
|
8
|
+
# not blocks, so they're callable from any Ractor.
|
|
9
|
+
module Worker
|
|
10
|
+
# Failures that retrying can't fix.
|
|
11
|
+
NEVER_RETRY = [UnknownJobError, NotImplementedError].freeze
|
|
12
|
+
|
|
13
|
+
# The body of a worker Ractor, until the supervisor sends :stop on the
|
|
14
|
+
# control port handed back through `hello`. A stop is only noticed
|
|
15
|
+
# between jobs, never in the middle of one. `holding` tells the
|
|
16
|
+
# supervisor which job this worker has, [index, id] after a claim and
|
|
17
|
+
# [index, nil] once it's finished or failed, so a job whose worker
|
|
18
|
+
# dies in between can be released.
|
|
19
|
+
def self.run(adapter, process_id, queues, poll_interval, hello, holding, index)
|
|
20
|
+
# The supervisor logs why a worker ended, backtrace included; don't
|
|
21
|
+
# also dump it to stderr.
|
|
22
|
+
Thread.current.report_on_exception = false
|
|
23
|
+
control = Ractor::Port.new
|
|
24
|
+
hello << control
|
|
25
|
+
wake = Thread::Queue.new
|
|
26
|
+
stopping = false
|
|
27
|
+
Thread.new do
|
|
28
|
+
control.receive
|
|
29
|
+
stopping = true
|
|
30
|
+
wake << :stop
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
until stopping
|
|
34
|
+
claim = next_claim(adapter, queues, process_id)
|
|
35
|
+
next wake.pop(timeout: poll_interval) unless claim
|
|
36
|
+
|
|
37
|
+
holding << [index, claim.id]
|
|
38
|
+
perform(adapter, process_id, claim)
|
|
39
|
+
holding << [index, nil]
|
|
40
|
+
end
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
def self.next_claim(adapter, queues, process_id)
|
|
44
|
+
queues.each do |queue|
|
|
45
|
+
claim = adapter.claim(queue, process_id)
|
|
46
|
+
return claim if claim
|
|
47
|
+
end
|
|
48
|
+
nil
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
# Every exception fails the job, not only StandardErrors: one that
|
|
52
|
+
# escaped would leave the job running inside a live process, which
|
|
53
|
+
# pruning never touches. A non-StandardError is re-raised afterwards,
|
|
54
|
+
# ending this Ractor so the supervisor starts a fresh one.
|
|
55
|
+
def self.perform(adapter, process_id, claim)
|
|
56
|
+
job = nil
|
|
57
|
+
error = begin
|
|
58
|
+
job = Monk::Jobs.lookup(claim.job_class)
|
|
59
|
+
run_job(job, claim.args)
|
|
60
|
+
nil
|
|
61
|
+
rescue Exception => e # rubocop:disable Lint/RescueException
|
|
62
|
+
e
|
|
63
|
+
end
|
|
64
|
+
return adapter.finish(claim.id, process_id) unless error
|
|
65
|
+
|
|
66
|
+
record_failure(adapter, process_id, claim, error, job)
|
|
67
|
+
raise error unless error.is_a?(StandardError) || error.is_a?(NotImplementedError)
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
def self.run_job(job, args)
|
|
71
|
+
return job.perform(*args) unless job.timeout
|
|
72
|
+
|
|
73
|
+
Timeout.timeout(job.timeout) { job.perform(*args) }
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
# job is nil when its class couldn't be found (UnknownJobError).
|
|
77
|
+
def self.record_failure(adapter, process_id, claim, error, job)
|
|
78
|
+
# Phase 0.3: the interrupted query is still running on the server,
|
|
79
|
+
# and this connection's next query would wait for it.
|
|
80
|
+
adapter.reset_connection if error.is_a?(Timeout::Error)
|
|
81
|
+
|
|
82
|
+
never_retry = NEVER_RETRY + (job ? job.never_retry : [])
|
|
83
|
+
retry_in = never_retry.any? { |klass| error.is_a?(klass) } ? nil : Monk::Jobs.backoff(claim.attempts)
|
|
84
|
+
outcome = adapter.fail(claim.id, process_id, error: Monk::Jobs.describe_error(error), retry_in: retry_in)
|
|
85
|
+
outcomes = { scheduled: "retrying in #{retry_in}s", failed: "failed for good" }
|
|
86
|
+
what_next = outcomes.fetch(outcome, "no longer held")
|
|
87
|
+
Monk::Log.error(
|
|
88
|
+
"Monk::Jobs: #{claim.job_class} (job #{claim.id}, attempt #{claim.attempts}) failed with " \
|
|
89
|
+
"#{error.class}: #{error.message} -- #{what_next}",
|
|
90
|
+
)
|
|
91
|
+
end
|
|
92
|
+
end
|
|
93
|
+
end
|
|
94
|
+
end
|
data/lib/monk/jobs.rb
ADDED
|
@@ -0,0 +1,229 @@
|
|
|
1
|
+
# The whole framework, as monk/live does: jobs use Monk.freeze!, Monk::Log
|
|
2
|
+
# and Monk.env, and a job process (bin/jobs) may load nothing else first.
|
|
3
|
+
require_relative "../monk"
|
|
4
|
+
require_relative "jobs/errors"
|
|
5
|
+
require_relative "jobs/args"
|
|
6
|
+
require "json"
|
|
7
|
+
|
|
8
|
+
module Monk
|
|
9
|
+
# Background jobs on Postgres. Opt-in: require "monk/jobs" explicitly --
|
|
10
|
+
# `require "monk"` alone does not load this. Design:
|
|
11
|
+
# docs/adr/0013-jobs-narrow-state-table-plus-payloads.md; plan:
|
|
12
|
+
# docs/history/plan-jobs.md.
|
|
13
|
+
module Jobs
|
|
14
|
+
# last_error keeps the first lines of a failure, not all of it.
|
|
15
|
+
MAX_ERROR_LENGTH = 4_000
|
|
16
|
+
BACKTRACE_LINES = 10
|
|
17
|
+
|
|
18
|
+
# The process id drain! claims jobs as: never a monk_processes id, which
|
|
19
|
+
# starts at 1.
|
|
20
|
+
DRAIN_PROCESS_ID = 0
|
|
21
|
+
|
|
22
|
+
class << self
|
|
23
|
+
# Typically, in config/jobs.rb:
|
|
24
|
+
#
|
|
25
|
+
# Monk::Jobs.configure(db_name: :primary)
|
|
26
|
+
#
|
|
27
|
+
# db_name: is a database registered with Monk::Persistence::Pg. The
|
|
28
|
+
# app's own database by default, so a job can be enqueued inside the
|
|
29
|
+
# app's own transaction; a separate one isolates the queue from the
|
|
30
|
+
# app's long transactions (docs/adr/0013-jobs-narrow-state-table-plus-payloads.md).
|
|
31
|
+
# Tests use the app's test database, the same way: enqueue, drain!,
|
|
32
|
+
# and clear! between tests.
|
|
33
|
+
def configure(db_name:)
|
|
34
|
+
require_relative "jobs/adapters/pg"
|
|
35
|
+
@adapter = Adapters::Pg.new(db_name: db_name)
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
# The configured adapter. Shareable (it freezes itself), so a worker
|
|
39
|
+
# Ractor reads it straight off this module.
|
|
40
|
+
def adapter
|
|
41
|
+
@adapter || raise(NotConfiguredError, "call Monk::Jobs.configure(db_name:) before enqueueing or running jobs")
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
# Stores a job to run later; returns its id. args are passed to the
|
|
45
|
+
# job's self.perform as they are, and must be plain JSON values
|
|
46
|
+
# (Args.check!). Options:
|
|
47
|
+
#
|
|
48
|
+
# wait: 60 run no sooner than 60 seconds from now
|
|
49
|
+
# at: Time run no sooner than then
|
|
50
|
+
# conn: a PG::Connection enqueue on it, inside its transaction, so
|
|
51
|
+
# the job commits or rolls back with the
|
|
52
|
+
# app's own writes
|
|
53
|
+
#
|
|
54
|
+
# SendReceipt.enqueue(order_id) is the same call.
|
|
55
|
+
def enqueue(job_class, *args, wait: nil, at: nil, conn: nil)
|
|
56
|
+
check_enqueueable!(job_class)
|
|
57
|
+
Args.check!(args)
|
|
58
|
+
check_schedule!(wait, at)
|
|
59
|
+
|
|
60
|
+
adapter.enqueue(
|
|
61
|
+
job_class: job_class.name, queue: job_class.queue, priority: job_class.priority,
|
|
62
|
+
max_attempts: job_class.max_attempts, args: JSON.generate(args), wait: wait, at: at, conn: conn,
|
|
63
|
+
)
|
|
64
|
+
end
|
|
65
|
+
|
|
66
|
+
# Called by Monk::Job.inherited, in the main Ractor as the app's job
|
|
67
|
+
# files load. Only recorded here: names are resolved when the
|
|
68
|
+
# registry is frozen, so a class that gets its constant name after
|
|
69
|
+
# being created (Foo = Class.new(Monk::Job)) is still found.
|
|
70
|
+
def record(job_class)
|
|
71
|
+
classes << job_class
|
|
72
|
+
end
|
|
73
|
+
|
|
74
|
+
# The Monk::Job subclass a monk_job_payloads.job_class names. Only
|
|
75
|
+
# ever answers with a recorded job class -- never Object.const_get
|
|
76
|
+
# on a String read from the database. Reads nothing but the frozen
|
|
77
|
+
# registry, so it works the same from any Ractor.
|
|
78
|
+
def lookup(name)
|
|
79
|
+
registry = @registry
|
|
80
|
+
if registry.nil?
|
|
81
|
+
raise NotFrozenError,
|
|
82
|
+
"Monk::Jobs.lookup(#{name.inspect}) before the job registry is frozen -- call Monk.boot(app) " \
|
|
83
|
+
"(or Monk.freeze!) in the main Ractor first, after the app's job classes are loaded"
|
|
84
|
+
end
|
|
85
|
+
|
|
86
|
+
registry.fetch(name) do
|
|
87
|
+
raise UnknownJobError, "no Monk::Job subclass named #{name.inspect} is loaded in this process"
|
|
88
|
+
end
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
# Called from Monk.freeze! via Monk.freeze_hooks. Rebuilds the map
|
|
92
|
+
# from every named job class recorded so far, so freezing twice (two
|
|
93
|
+
# apps, or a test suite) is harmless. Anonymous classes are skipped:
|
|
94
|
+
# without a name a job can never be found again once enqueued.
|
|
95
|
+
def freeze_registry!
|
|
96
|
+
@registry = Ractor.make_shareable(classes.filter_map { |job| [job.name, job] if job.name }.to_h)
|
|
97
|
+
end
|
|
98
|
+
|
|
99
|
+
# A failed job back to available, with its attempts reset; true if it
|
|
100
|
+
# was failed. Failed jobs stay in monk_jobs until retried or discarded.
|
|
101
|
+
def retry_failed(id)
|
|
102
|
+
adapter.retry_failed(id)
|
|
103
|
+
end
|
|
104
|
+
|
|
105
|
+
# Deletes a failed job; true if it was failed.
|
|
106
|
+
def discard_failed(id)
|
|
107
|
+
adapter.discard_failed(id)
|
|
108
|
+
end
|
|
109
|
+
|
|
110
|
+
# Seconds to wait before retrying a job that has failed `attempts`
|
|
111
|
+
# times: 16, 31, 96, 271, 640, ... (Sidekiq's curve, without its
|
|
112
|
+
# random jitter).
|
|
113
|
+
def backoff(attempts)
|
|
114
|
+
(attempts**4) + 15
|
|
115
|
+
end
|
|
116
|
+
|
|
117
|
+
# What goes in last_error: the class and message, then the first
|
|
118
|
+
# BACKTRACE_LINES of the backtrace, capped at MAX_ERROR_LENGTH.
|
|
119
|
+
def describe_error(error)
|
|
120
|
+
lines = ["#{error.class}: #{error.message}", *Array(error.backtrace).first(BACKTRACE_LINES)]
|
|
121
|
+
lines.join("\n")[0, MAX_ERROR_LENGTH]
|
|
122
|
+
end
|
|
123
|
+
|
|
124
|
+
# Runs every job that's due, one after another, until none is left --
|
|
125
|
+
# including jobs those jobs enqueue -- and returns how many ran. For
|
|
126
|
+
# an app's tests: enqueue, drain!, then assert on what the jobs did.
|
|
127
|
+
# Nothing is retried: the first job that raises is marked failed and
|
|
128
|
+
# its error re-raised here.
|
|
129
|
+
#
|
|
130
|
+
# Each job runs in a throwaway Ractor, as it would in bin/jobs, so a
|
|
131
|
+
# job that only works in the main Ractor (a gem with module-level
|
|
132
|
+
# state, a non-shareable constant) fails in the test too.
|
|
133
|
+
# in_ractor: false runs jobs in the calling Ractor instead, for a test
|
|
134
|
+
# that needs to see a job's side effects there.
|
|
135
|
+
#
|
|
136
|
+
# Calls Monk.freeze! first, as bin/jobs does, so jobs can read the
|
|
137
|
+
# app's frozen configuration from their Ractor.
|
|
138
|
+
def drain!(in_ractor: true)
|
|
139
|
+
Monk.freeze!
|
|
140
|
+
queues = classes.map(&:queue) | [Job::DEFAULT_QUEUE]
|
|
141
|
+
ran = 0
|
|
142
|
+
loop do
|
|
143
|
+
nil while adapter.stage_due.positive?
|
|
144
|
+
claim = queues.lazy.filter_map { |queue| adapter.claim(queue, DRAIN_PROCESS_ID) }.first
|
|
145
|
+
break unless claim
|
|
146
|
+
|
|
147
|
+
run_drained(claim, in_ractor)
|
|
148
|
+
ran += 1
|
|
149
|
+
end
|
|
150
|
+
ran
|
|
151
|
+
end
|
|
152
|
+
|
|
153
|
+
# Empties the queue: every job, failed ones included, and every job
|
|
154
|
+
# process row. For an app's tests, between one test and the next --
|
|
155
|
+
# and so it refuses to run unless MONK_ENV is test.
|
|
156
|
+
def clear!
|
|
157
|
+
unless Monk.env.test?
|
|
158
|
+
raise ClearOutsideTestsError,
|
|
159
|
+
"Monk::Jobs.clear! empties the whole queue, so it only runs with MONK_ENV=test (this is #{Monk.env})"
|
|
160
|
+
end
|
|
161
|
+
|
|
162
|
+
adapter.clear!
|
|
163
|
+
end
|
|
164
|
+
|
|
165
|
+
# Test-only: unfreezes the registry and forgets the configuration.
|
|
166
|
+
# Recorded classes stay recorded, the same way a Ruby class can't be
|
|
167
|
+
# undefined.
|
|
168
|
+
def reset!
|
|
169
|
+
@registry = nil
|
|
170
|
+
@adapter = nil
|
|
171
|
+
end
|
|
172
|
+
|
|
173
|
+
private
|
|
174
|
+
|
|
175
|
+
def run_drained(claim, in_ractor)
|
|
176
|
+
begin
|
|
177
|
+
job = lookup(claim.job_class)
|
|
178
|
+
in_ractor ? perform_in_ractor(job, claim.args) : job.perform(*claim.args)
|
|
179
|
+
# NotImplementedError too, a job class with no self.perform: it's a
|
|
180
|
+
# ScriptError, not a StandardError, and would otherwise leave the job
|
|
181
|
+
# running.
|
|
182
|
+
rescue StandardError, NotImplementedError => e
|
|
183
|
+
adapter.fail(claim.id, DRAIN_PROCESS_ID, error: describe_error(e), retry_in: nil)
|
|
184
|
+
raise
|
|
185
|
+
end
|
|
186
|
+
adapter.finish(claim.id, DRAIN_PROCESS_ID)
|
|
187
|
+
end
|
|
188
|
+
|
|
189
|
+
def perform_in_ractor(job, args)
|
|
190
|
+
Ractor.new(job, args) do |job, args|
|
|
191
|
+
# The error is re-raised in the caller; don't also print it here.
|
|
192
|
+
Thread.current.report_on_exception = false
|
|
193
|
+
job.perform(*args)
|
|
194
|
+
nil
|
|
195
|
+
end.value
|
|
196
|
+
rescue Ractor::RemoteError => e
|
|
197
|
+
# The job's own error, as if it had run here. cause: nil, or Ruby
|
|
198
|
+
# would chain the RemoteError onto it -- whose cause it already is.
|
|
199
|
+
raise e.cause, cause: nil
|
|
200
|
+
end
|
|
201
|
+
|
|
202
|
+
def check_enqueueable!(job_class)
|
|
203
|
+
unless job_class.is_a?(Class) && job_class < Monk::Job
|
|
204
|
+
raise ArgumentError, "Monk::Jobs.enqueue needs a Monk::Job subclass, got #{job_class.inspect}"
|
|
205
|
+
end
|
|
206
|
+
return if job_class.name
|
|
207
|
+
|
|
208
|
+
raise ArgumentError,
|
|
209
|
+
"#{job_class.inspect} has no name, so a worker could never find it again -- assign it to a constant"
|
|
210
|
+
end
|
|
211
|
+
|
|
212
|
+
def check_schedule!(wait, at)
|
|
213
|
+
raise ArgumentError, "pass wait: or at:, not both" if wait && at
|
|
214
|
+
if wait && !(wait.is_a?(Numeric) && wait >= 0)
|
|
215
|
+
raise ArgumentError, "wait: must be a number of seconds, zero or more, got #{wait.inspect}"
|
|
216
|
+
end
|
|
217
|
+
raise ArgumentError, "at: must be a Time, got #{at.inspect}" if at && !at.is_a?(Time)
|
|
218
|
+
end
|
|
219
|
+
|
|
220
|
+
def classes
|
|
221
|
+
@classes ||= []
|
|
222
|
+
end
|
|
223
|
+
end
|
|
224
|
+
|
|
225
|
+
Monk.freeze_hooks << self
|
|
226
|
+
end
|
|
227
|
+
end
|
|
228
|
+
|
|
229
|
+
require_relative "jobs/job"
|
data/lib/monk/mail/errors.rb
CHANGED
|
@@ -24,5 +24,12 @@ module Monk
|
|
|
24
24
|
# refusing it. The transport's own exception is the #cause.
|
|
25
25
|
class DeliveryError < StandardError
|
|
26
26
|
end
|
|
27
|
+
|
|
28
|
+
# A DeliveryError the server made final: an SMTP 5xx (no such mailbox,
|
|
29
|
+
# relaying refused) or refused credentials. Raised by
|
|
30
|
+
# Monk::Mail::DeliveryJob (require "monk/mail/later") so the job isn't
|
|
31
|
+
# retried; still a DeliveryError, so a rescue of that catches it too.
|
|
32
|
+
class PermanentDeliveryError < DeliveryError
|
|
33
|
+
end
|
|
27
34
|
end
|
|
28
35
|
end
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
require_relative "../mail"
|
|
2
|
+
require_relative "../jobs"
|
|
3
|
+
|
|
4
|
+
module Monk
|
|
5
|
+
module Mail
|
|
6
|
+
# Sends one message from a job process (docs/adr/0014-mail-from-jobs-and-login-links-created-in-the-job.md).
|
|
7
|
+
# Enqueued by Monk::Mail.deliver_later, never directly: its argument
|
|
8
|
+
# is the message's fields, already checked and with from: resolved.
|
|
9
|
+
class DeliveryJob < Monk::Job
|
|
10
|
+
queue "mailers"
|
|
11
|
+
never_retry PermanentDeliveryError
|
|
12
|
+
|
|
13
|
+
# The SMTP failures that are final, by class name so this file never
|
|
14
|
+
# loads net/smtp itself (the app's Gemfile decides that, ADR 0012):
|
|
15
|
+
# a 5xx reply, a malformed command, refused credentials. Everything
|
|
16
|
+
# else -- a 4xx "try later", a refused connection, TLS, a timeout --
|
|
17
|
+
# is worth retrying.
|
|
18
|
+
PERMANENT_SMTP_ERRORS = %w[Net::SMTPFatalError Net::SMTPSyntaxError Net::SMTPAuthenticationError].freeze
|
|
19
|
+
|
|
20
|
+
def self.perform(fields)
|
|
21
|
+
Monk::Mail.deliver(**fields.transform_keys(&:to_sym))
|
|
22
|
+
rescue DeliveryError => e
|
|
23
|
+
raise unless PERMANENT_SMTP_ERRORS.include?(e.cause&.class&.name)
|
|
24
|
+
|
|
25
|
+
raise PermanentDeliveryError, e.message
|
|
26
|
+
end
|
|
27
|
+
end
|
|
28
|
+
|
|
29
|
+
class << self
|
|
30
|
+
# deliver, later: the same arguments, plus enqueue's wait:, at: and
|
|
31
|
+
# conn: (the last one enqueues inside the caller's transaction, so the
|
|
32
|
+
# email commits or rolls back with the app's own writes). Returns the
|
|
33
|
+
# job's id. The message is built -- and so checked -- here, before
|
|
34
|
+
# anything is enqueued: bad input raises InvalidMessageError in the
|
|
35
|
+
# caller, exactly as deliver would, rather than failing later in the
|
|
36
|
+
# job process. Render HTML before calling (html: Monk::Mail.render(...)):
|
|
37
|
+
# the job stores the finished message and only sends it.
|
|
38
|
+
#
|
|
39
|
+
# Monk::Mail.deliver_later(to: order[:email], subject: "Your receipt", text: receipt_text)
|
|
40
|
+
def deliver_later(to:, subject:, text: nil, html: nil, reply_to: nil, from: nil, wait: nil, at: nil, conn: nil)
|
|
41
|
+
raise NotConfiguredError, "call Monk::Mail.configure(url:, from:) before Monk::Mail.deliver_later" unless config
|
|
42
|
+
|
|
43
|
+
from ||= config[:from]
|
|
44
|
+
if from.nil?
|
|
45
|
+
raise InvalidMessageError, "no from: given, and Monk::Mail.configure has no default from: (MAIL_FROM)"
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
message = Message.new(from: from, to: to, subject: subject, text: text, html: html, reply_to: reply_to)
|
|
49
|
+
DeliveryJob.enqueue(message.to_h.transform_keys(&:to_s), wait: wait, at: at, conn: conn)
|
|
50
|
+
end
|
|
51
|
+
end
|
|
52
|
+
end
|
|
53
|
+
end
|
data/lib/monk/persistence/pg.rb
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
require "json"
|
|
1
2
|
require "pg"
|
|
2
3
|
|
|
3
4
|
require_relative "../persistence"
|
|
@@ -16,12 +17,31 @@ module Monk
|
|
|
16
17
|
module Pg
|
|
17
18
|
extend Monk::Persistence::Registry
|
|
18
19
|
|
|
20
|
+
# json/jsonb columns, decoded with plain JSON.parse. pg's own
|
|
21
|
+
# PG::TextDecoder::JSON (1.6.3) calls JSON.parse(string, quirks_mode:
|
|
22
|
+
# true), and json 3 removed that keyword, so with pg's default every
|
|
23
|
+
# json/jsonb read raised ArgumentError. json 3 parses a bare scalar
|
|
24
|
+
# ("7", "\"text\"") without it.
|
|
25
|
+
class JsonDecoder < PG::SimpleDecoder
|
|
26
|
+
def decode(string, _tuple = nil, _field = nil)
|
|
27
|
+
JSON.parse(string)
|
|
28
|
+
end
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
# pg's default result types, with JsonDecoder for json and jsonb.
|
|
32
|
+
# Built once and made shareable, so every Ractor's #connect can use
|
|
33
|
+
# it rather than building its own.
|
|
34
|
+
RESULT_TYPES = Ractor.make_shareable(
|
|
35
|
+
PG::BasicTypeRegistry.new.register_default_types
|
|
36
|
+
.tap { |types| %w[json jsonb].each { |name| types.register_type(0, name, nil, JsonDecoder) } },
|
|
37
|
+
)
|
|
38
|
+
|
|
19
39
|
class << self
|
|
20
40
|
private
|
|
21
41
|
|
|
22
42
|
def connect(**opts)
|
|
23
43
|
conn = PG.connect(**opts)
|
|
24
|
-
conn.type_map_for_results = PG::BasicTypeMapForResults.new(conn)
|
|
44
|
+
conn.type_map_for_results = PG::BasicTypeMapForResults.new(conn, registry: RESULT_TYPES)
|
|
25
45
|
conn
|
|
26
46
|
end
|
|
27
47
|
|