riverqueue 0.12.0 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -6,6 +6,21 @@ module River
6
6
  class ClientRuntime
7
7
  class Interrupted < StandardError; end
8
8
 
9
+ # Keep wall time for queue latency and error timestamps, and monotonic time
10
+ # for execution and completion durations. Externally executed jobs start
11
+ # with no locally observed execution time. Snapshot queue latency before
12
+ # retries, snoozes, or application code change the job's schedule.
13
+ class JobTiming
14
+ attr_accessor :finished_monotonic
15
+ attr_reader :queue_wait_duration, :started_at, :started_monotonic
16
+
17
+ def initialize(started_at, started_monotonic, scheduled_at)
18
+ @queue_wait_duration = started_at - scheduled_at
19
+ @started_at = started_at
20
+ @finished_monotonic = @started_monotonic = started_monotonic
21
+ end
22
+ end
23
+
9
24
  attr_reader :periodic_jobs
10
25
 
11
26
  def initialize(client, driver, config)
@@ -14,7 +29,6 @@ module River
14
29
  @config = config
15
30
  @driver = driver
16
31
  @mutex = Mutex.new
17
- @periodic_jobs = PeriodicJobBundle.new(config.periodic_jobs, wake: method(:wake))
18
32
  @producer_threads = {}
19
33
  @queue_configs = config.queues.dup
20
34
  @removed_queues = {}
@@ -23,17 +37,37 @@ module River
23
37
  @stopped = true
24
38
  @subscriptions = []
25
39
  @threads = []
40
+ @periodic_jobs = PeriodicJobBundle.new(config.periodic_jobs,
41
+ enabled: !config.leader_election_disabled, wake: method(:periodic_jobs_changed))
26
42
  end
27
43
 
28
44
  def finish_claimed(row, error = nil)
29
- started_at = Time.now.utc
45
+ timing = JobTiming.new(row.attempted_at || Time.now.utc, monotonic_now, row.scheduled_at)
30
46
  job = Job.new(@client, row)
31
- if error
32
- finish_failed(row, job, error, started_at)
33
- else
47
+ case error
48
+ when nil
34
49
  now = Time.now.utc
35
- completed = @driver.job_set_state_if_running(id: row.id, finalized_at: now, now: now, state: JOB_STATE_COMPLETED)
36
- publish(EVENT_JOB_COMPLETED, completed, started_at) if completed
50
+ completed = @driver.job_complete(id: row.id, finalized_at: now, now: now)
51
+ case completed
52
+ when :cancelled
53
+ finish_failed(row, job, JobCancelError.new, timing, cancelled: true)
54
+ else
55
+ publish(EVENT_JOB_COMPLETED, completed, timing) if completed
56
+ end
57
+ when JobCancelError
58
+ finish_failed(row, job, error, timing, cancelled: true)
59
+ when JobSnoozeError
60
+ finish_snoozed(row, job, error, timing)
61
+ when Interrupted
62
+ finish_interrupted(row, job, timing)
63
+ else
64
+ worker = begin
65
+ resolve_worker(row.kind) if !row.__decode_error && @config.workers.include?(row.kind)
66
+ rescue => worker_error
67
+ @config.logger.error("River worker initialization failed during finalization: #{worker_error.full_message}")
68
+ nil
69
+ end
70
+ finish_failed(row, job, error, timing, worker: worker)
37
71
  end
38
72
  end
39
73
 
@@ -115,6 +149,9 @@ module River
115
149
  name = name.to_s
116
150
  producer = @mutex.synchronize do
117
151
  raise NotFoundError, "queue is not configured: #{name}" unless @queue_configs.key?(name)
152
+ if @producer_threads[name] == Thread.current || @running.any? { |_id, entry| entry[:queue] == name && entry[:thread] == Thread.current }
153
+ raise ThreadError, "cannot remove a queue from its own worker or producer"
154
+ end
118
155
 
119
156
  @removed_queues[name] = true
120
157
  @condition.broadcast
@@ -148,7 +185,7 @@ module River
148
185
  begin
149
186
  queues = @mutex.synchronize { @queue_configs.dup }
150
187
  queues.each { |name, queue_config| start_producer(name, queue_config) }
151
- start_maintenance unless @queue_configs.empty?
188
+ start_maintenance unless @queue_configs.empty? && @periodic_jobs.empty? && @config.maintenance_services.empty?
152
189
 
153
190
  self
154
191
  rescue
@@ -164,6 +201,9 @@ module River
164
201
  def stop(cancel: false, wait: true)
165
202
  threads = @mutex.synchronize do
166
203
  return self if @stopped
204
+ if wait && (@threads.include?(Thread.current) || @running.any? { |_id, entry| entry[:thread] == Thread.current })
205
+ raise ThreadError, "cannot wait for client stop from its own runtime thread; use wait: false"
206
+ end
167
207
 
168
208
  @stop_requested = true
169
209
  @condition.broadcast
@@ -246,18 +286,20 @@ module River
246
286
  end
247
287
 
248
288
  private def execute(row)
249
- started_at = Time.now.utc
289
+ timing = JobTiming.new(Time.now.utc, monotonic_now, row.scheduled_at)
250
290
  job = Job.new(@client, row)
291
+ worker = nil #: untyped
251
292
  begin
252
- raise row.__decode_error if row.__decode_error
253
-
254
- worker = resolve_worker(row.kind)
255
- begin_work(row.id)
256
293
  begin
294
+ raise row.__decode_error if row.__decode_error
295
+
296
+ worker = resolve_worker(row.kind)
297
+ begin_work(row.id)
257
298
  Thread.handle_interrupt(Interrupted => :immediate, JobCancelError => :immediate) do
258
299
  invoke_worker(worker, job)
259
300
  end
260
301
  ensure
302
+ timing.finished_monotonic = monotonic_now
261
303
  finish_work(row.id)
262
304
  end
263
305
 
@@ -266,46 +308,33 @@ module River
266
308
  raise JobCancelError if @driver.job_get_cancelled_ids([row.id]).include?(row.id)
267
309
 
268
310
  if invoke_plugins(:job_finalize, job, JOB_STATE_COMPLETED).include?(:delete)
269
- @driver.job_delete_if_running(row.id)
311
+ raise JobCancelError if @driver.job_delete_if_running(row.id) == :cancelled
312
+
270
313
  return [:deleted, nil]
271
314
  end
272
315
  end
273
316
 
274
317
  completed_at = Time.now.utc
275
- completed = if finalize_hooks
276
- @driver.job_set_state_if_running(id: row.id, finalized_at: completed_at, metadata: job.metadata_updates, now: completed_at, state: JOB_STATE_COMPLETED)
277
- else
278
- result = @driver.job_complete(id: row.id, finalized_at: completed_at, metadata: job.metadata_updates, now: completed_at)
279
- case result
280
- when :cancelled then raise JobCancelError
281
- else result
282
- end
318
+ completed = @driver.job_complete(id: row.id, finalized_at: completed_at, metadata: job.metadata_updates, now: completed_at)
319
+ case completed
320
+ when :cancelled then raise JobCancelError
321
+ else publish(EVENT_JOB_COMPLETED, completed, timing) if completed
283
322
  end
284
- publish(EVENT_JOB_COMPLETED, completed, started_at) if completed
285
323
 
286
324
  [:completed, nil]
287
325
  rescue JobSnoozeError => error
288
326
  job.__capture_resumable_metadata!
289
- [finish_snoozed(row, job, error, started_at), error]
327
+ [finish_snoozed(row, job, error, timing), error]
290
328
  rescue JobCancelError => error
291
329
  job.__capture_resumable_metadata!
292
- finish_failed(row, job, error, started_at, cancelled: true)
330
+ finish_failed(row, job, error, timing, cancelled: true)
293
331
  [:cancelled, error]
294
332
  rescue Interrupted => error
295
333
  job.__capture_resumable_metadata!
296
- interrupted = @driver.job_set_state_if_running(
297
- id: row.id,
298
- attempt: [row.attempt - 1, 0].max,
299
- metadata: job.metadata_updates,
300
- scheduled_at: Time.now.utc,
301
- state: JOB_STATE_AVAILABLE
302
- )
303
- publish(EVENT_JOB_INTERRUPTED, interrupted, started_at) if interrupted
304
-
305
- [(interrupted&.state == JOB_STATE_CANCELLED) ? :cancelled : :interrupted, error]
334
+ [finish_interrupted(row, job, timing), error]
306
335
  rescue => error
307
336
  job.__capture_resumable_metadata!
308
- [finish_failed(row, job, error, started_at, worker: worker), error]
337
+ [finish_failed(row, job, error, timing, worker: worker), error]
309
338
  ensure
310
339
  @mutex.synchronize do
311
340
  @running.delete(row.id)
@@ -314,11 +343,11 @@ module River
314
343
  end
315
344
  end
316
345
 
317
- private def finish_failed(row, job, error, started_at, cancelled: false, worker: nil)
346
+ private def finish_failed(row, job, error, timing, cancelled: false, worker: nil)
318
347
  cancelled ||= error_handler_cancel?(error, job)
319
348
  now = Time.now.utc
320
349
  attempt_error = AttemptError.new(
321
- at: started_at,
350
+ at: timing.started_at,
322
351
  attempt: row.attempt,
323
352
  error: error.message,
324
353
  trace: Array(error.backtrace).join("\n")
@@ -344,7 +373,7 @@ module River
344
373
  state: state
345
374
  )
346
375
  event = cancelled ? EVENT_JOB_CANCELLED : EVENT_JOB_FAILED
347
- publish(event, updated, started_at) if updated
376
+ publish(event, updated, timing) if updated
348
377
 
349
378
  case updated&.state || state
350
379
  when JOB_STATE_CANCELLED then :cancelled
@@ -353,10 +382,23 @@ module River
353
382
  end
354
383
  end
355
384
 
356
- private def finish_snoozed(row, job, error, started_at)
385
+ private def finish_interrupted(row, job, timing)
386
+ interrupted = @driver.job_set_state_if_running(
387
+ id: row.id,
388
+ attempt: [row.attempt - 1, 0].max,
389
+ metadata: job.metadata_updates,
390
+ scheduled_at: Time.now.utc,
391
+ state: JOB_STATE_AVAILABLE
392
+ )
393
+ publish(EVENT_JOB_INTERRUPTED, interrupted, timing) if interrupted
394
+
395
+ (interrupted&.state == JOB_STATE_CANCELLED) ? :cancelled : :interrupted
396
+ end
397
+
398
+ private def finish_snoozed(row, job, error, timing)
357
399
  scheduled_at = Time.now.utc + error.duration
358
400
  state = (error.duration <= 5) ? JOB_STATE_AVAILABLE : JOB_STATE_SCHEDULED
359
- metadata = job.metadata_updates.merge("snoozes" => row.metadata.fetch("snoozes", 0).to_i + 1)
401
+ metadata = job.metadata_updates.merge("snoozes" => next_snooze_count(row.metadata["snoozes"]))
360
402
  updated = @driver.job_set_state_if_running(
361
403
  id: row.id,
362
404
  attempt: [row.attempt - 1, 0].max,
@@ -364,7 +406,7 @@ module River
364
406
  scheduled_at: scheduled_at,
365
407
  state: state
366
408
  )
367
- publish(EVENT_JOB_SNOOZED, updated, started_at) if updated
409
+ publish(EVENT_JOB_SNOOZED, updated, timing) if updated
368
410
  (updated&.state == JOB_STATE_CANCELLED) ? :cancelled : :snoozed
369
411
  end
370
412
 
@@ -429,12 +471,17 @@ module River
429
471
  if now >= next_schedule
430
472
  @driver.job_schedule(now: now)
431
473
  run_periodic(now)
432
- @config.maintenance_services.each { |service| service.run(@client, @driver, now) }
474
+ @config.maintenance_services.each do |service|
475
+ service.run(@client, @driver, now)
476
+ rescue => error
477
+ @config.logger.error("River maintenance service failed (#{service.class}): #{error.full_message}")
478
+ end
433
479
  next_schedule = now + 5
434
480
  end
435
481
 
436
482
  if now >= next_rescue
437
- @driver.job_rescue_stuck(horizon: now - 3_600, now: now, retry_policy: @config.retry_policy)
483
+ @driver.job_rescue_stuck(horizon: now - 3_600, logger: @config.logger, now: now,
484
+ rescue_if: method(:rescue_job?), retry_policy: @config.retry_policy)
438
485
  next_rescue = now + 30
439
486
  end
440
487
 
@@ -474,17 +521,33 @@ module River
474
521
  DefaultClientRetryPolicy.new.next_retry(row, error, now: now)
475
522
  end
476
523
 
477
- private def perform_work(worker, job)
478
- timeout = worker.respond_to?(:timeout) ? worker.timeout(job) : @config.job_timeout
479
- timeout = Float(timeout) unless timeout.nil?
480
- timeout = @config.job_timeout if timeout == 0
481
- raise ArgumentError, "worker timeout must be finite and nonnegative, or nil" if timeout && (!timeout.finite? || timeout.negative?)
524
+ # Preserve Ruby's lenient counter conversion without calling to_i on
525
+ # booleans/collections, or accepting only a prefix of a numeric string.
526
+ # Only canonical non-negative integers are part of the shared protocol.
527
+ private def next_snooze_count(value)
528
+ count = case value
529
+ when Integer then value
530
+ when Float then value.finite? ? value.to_i : 0
531
+ when String then /\A-?[0-9]+\z/.match?(value) ? value.to_i : 0
532
+ when true then 1
533
+ else 0
534
+ end
535
+ # Go stores and increments a signed 64-bit integer.
536
+ ((count + 1 + (1 << 63)) % (1 << 64)) - (1 << 63)
537
+ end
482
538
 
539
+ private def perform_work(worker, job)
540
+ timeout = worker_timeout(worker, job)
483
541
  result = timeout ? Timeout.timeout(timeout) { worker.work(job) } : worker.work(job)
484
542
  job.__finish_resumable_work!
485
543
  result
486
544
  end
487
545
 
546
+ private def periodic_jobs_changed
547
+ start_maintenance
548
+ wake
549
+ end
550
+
488
551
  private def producer_loop(queue, queue_config)
489
552
  cooldown = queue_config.resolved_fetch_cooldown(@config)
490
553
  # Capture registrations (including aliases) at startup, not construction.
@@ -493,9 +556,15 @@ module River
493
556
  last_fetch = 0.0
494
557
  poll_interval = queue_config.resolved_fetch_poll_interval(@config)
495
558
  loop do
496
- break if queue_stopping?(queue)
559
+ draining = queue_stopping?(queue)
560
+ break if draining && running_count(queue).zero?
497
561
 
498
562
  check_remote_cancellations(queue)
563
+ if draining
564
+ wait_for_running_jobs(queue, poll_interval)
565
+ next
566
+ end
567
+
499
568
  capacity = queue_config.max_workers - running_count(queue)
500
569
  if capacity.positive?
501
570
  # Inserts, completions, and spurious condition wakes must not bypass
@@ -503,7 +572,7 @@ module River
503
572
  while (sleep_for = cooldown - (monotonic_now - last_fetch)).positive? && !queue_stopping?(queue)
504
573
  wait(sleep_for)
505
574
  end
506
- break if queue_stopping?(queue)
575
+ next if queue_stopping?(queue)
507
576
 
508
577
  # A pause may arrive during cooldown; only read the queue immediately
509
578
  # before fetching, and avoid the read entirely when all slots are busy.
@@ -519,18 +588,25 @@ module River
519
588
  end
520
589
  rescue => error
521
590
  @config.logger.error("River producer for #{queue.inspect} stopped: #{error.full_message}")
522
- wait(poll_interval || @config.fetch_poll_interval)
523
- retry unless queue_stopping?(queue)
591
+ if queue_stopping?(queue)
592
+ wait_for_running_jobs(queue, poll_interval || @config.fetch_poll_interval)
593
+ else
594
+ wait(poll_interval || @config.fetch_poll_interval)
595
+ end
596
+ retry unless queue_stopping?(queue) && running_count(queue).zero?
524
597
  end
525
598
 
526
- private def publish(kind, job, started_at)
599
+ private def publish(kind, job, timing)
527
600
  # Another actor may have changed the row after our update. Pending,
528
601
  # running, and unknown states do not represent a settled attempt.
529
602
  return unless %w[available cancelled completed discarded retryable scheduled].include?(job.state)
530
603
 
531
604
  kind = EVENT_JOB_CANCELLED if job.state == JOB_STATE_CANCELLED
532
- completed_at = Time.now.utc
533
- stats = JobStatistics.new(0, started_at - job.scheduled_at, completed_at - started_at)
605
+ stats = JobStatistics.new(
606
+ monotonic_now - timing.finished_monotonic,
607
+ timing.queue_wait_duration,
608
+ timing.finished_monotonic - timing.started_monotonic
609
+ )
534
610
  event = Event.new(kind, job, nil, stats)
535
611
  @mutex.synchronize { @subscriptions.dup }.each { |subscription| subscription.publish(event) }
536
612
  end
@@ -543,6 +619,19 @@ module River
543
619
  @mutex.synchronize { @subscriptions.delete(subscription) }
544
620
  end
545
621
 
622
+ private def rescue_job?(row, now)
623
+ return true if row.__decode_error
624
+
625
+ worker = @config.workers[row.kind]
626
+ worker = worker.new if worker.is_a?(Class)
627
+ timeout = worker_timeout(worker, Job.new(@client, row))
628
+ attempted_at = row.attempted_at #: Time
629
+ !timeout.nil? && now - attempted_at >= timeout
630
+ rescue => error
631
+ @config.logger.error("River rescue timeout check failed; rescuing attempt: #{error.full_message}")
632
+ true
633
+ end
634
+
546
635
  private def resolve_worker(kind)
547
636
  worker = @config.workers[kind]
548
637
  raise UnknownJobKindError, kind unless worker
@@ -565,7 +654,10 @@ module River
565
654
  next unless value
566
655
 
567
656
  args, opts = value.is_a?(Array) ? value : [value, nil]
568
- @client.insert(args, insert_opts: opts || InsertOpts.new)
657
+ opts = (opts || InsertOpts.new).dup
658
+ opts.metadata = (opts.metadata || {}).merge("periodic" => true)
659
+ opts.metadata["river:periodic_job_id"] = periodic_job.id if periodic_job.id && !periodic_job.id.empty?
660
+ @client.insert(args, insert_opts: opts)
569
661
  rescue => error
570
662
  @config.logger.error("River periodic job failed to insert: #{error.full_message}")
571
663
  end
@@ -579,7 +671,7 @@ module River
579
671
  return if @config.leader_election_disabled
580
672
 
581
673
  @mutex.synchronize do
582
- return if @stop_requested
674
+ return unless @started && !@stop_requested
583
675
  return if @maintenance_thread&.alive?
584
676
 
585
677
  thread = Thread.new { maintenance_loop }
@@ -608,5 +700,22 @@ module River
608
700
  private def wait(duration)
609
701
  @mutex.synchronize { @condition.wait(@mutex, duration) unless @stop_requested }
610
702
  end
703
+
704
+ private def wait_for_running_jobs(queue, duration)
705
+ @mutex.synchronize do
706
+ # Draining still needs a poll interval after stop is requested. Check
707
+ # under the same mutex as completion so the last job's wake isn't lost.
708
+ @condition.wait(@mutex, duration) if @running.any? { |_id, entry| entry[:queue] == queue }
709
+ end
710
+ end
711
+
712
+ private def worker_timeout(worker, job)
713
+ timeout = worker.respond_to?(:timeout) ? worker.timeout(job) : @config.job_timeout
714
+ timeout = Float(timeout) unless timeout.nil?
715
+ timeout = @config.job_timeout if timeout == 0
716
+ raise ArgumentError, "worker timeout must be finite and nonnegative, or nil" if timeout && (!timeout.finite? || timeout.negative?)
717
+
718
+ timeout
719
+ end
611
720
  end
612
721
  end