wurk 1.1.0 → 1.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (113) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +25 -0
  3. data/app/controllers/concerns/wurk/same_origin_guard.rb +40 -0
  4. data/app/controllers/concerns/wurk/sse_streaming.rb +48 -0
  5. data/app/controllers/concerns/wurk/stream_concurrency_guard.rb +53 -0
  6. data/app/controllers/wurk/api/pagination.rb +60 -11
  7. data/app/controllers/wurk/api/serializers.rb +5 -1
  8. data/app/controllers/wurk/api_controller.rb +51 -70
  9. data/app/controllers/wurk/application_controller.rb +26 -0
  10. data/app/controllers/wurk/dashboard_controller.rb +23 -1
  11. data/app/controllers/wurk/extensions_controller.rb +3 -12
  12. data/app/controllers/wurk/profiles_controller.rb +6 -1
  13. data/lib/wurk/batch/death_handler.rb +13 -0
  14. data/lib/wurk/batch/server_middleware.rb +9 -5
  15. data/lib/wurk/batch.rb +3 -0
  16. data/lib/wurk/capsule.rb +34 -21
  17. data/lib/wurk/cli.rb +16 -3
  18. data/lib/wurk/client/buffered.rb +30 -10
  19. data/lib/wurk/client.rb +45 -1
  20. data/lib/wurk/component.rb +23 -9
  21. data/lib/wurk/configuration.rb +83 -18
  22. data/lib/wurk/cron.rb +13 -1
  23. data/lib/wurk/dead_set.rb +16 -1
  24. data/lib/wurk/engine.rb +17 -1
  25. data/lib/wurk/fetcher/reliable.rb +46 -13
  26. data/lib/wurk/fetcher.rb +5 -0
  27. data/lib/wurk/health.rb +74 -22
  28. data/lib/wurk/history.rb +4 -21
  29. data/lib/wurk/job_retry.rb +3 -5
  30. data/lib/wurk/launcher.rb +26 -3
  31. data/lib/wurk/limiter/server_middleware.rb +16 -1
  32. data/lib/wurk/limiter.rb +6 -5
  33. data/lib/wurk/lua/loader.rb +11 -5
  34. data/lib/wurk/lua.rb +100 -9
  35. data/lib/wurk/manager.rb +59 -19
  36. data/lib/wurk/metrics/history.rb +24 -38
  37. data/lib/wurk/metrics/queue_rollup.rb +4 -21
  38. data/lib/wurk/metrics/rollup.rb +4 -21
  39. data/lib/wurk/processor.rb +8 -8
  40. data/lib/wurk/profile_set.rb +21 -6
  41. data/lib/wurk/profiler.rb +7 -5
  42. data/lib/wurk/rails_boot.rb +176 -0
  43. data/lib/wurk/railtie.rb +19 -47
  44. data/lib/wurk/redis_connection.rb +6 -9
  45. data/lib/wurk/redis_pool.rb +148 -28
  46. data/lib/wurk/scheduled.rb +35 -12
  47. data/lib/wurk/swarm/backoff.rb +70 -0
  48. data/lib/wurk/swarm/child_boot.rb +92 -13
  49. data/lib/wurk/swarm/orphan_guard.rb +105 -0
  50. data/lib/wurk/swarm/restart.rb +196 -0
  51. data/lib/wurk/swarm.rb +194 -78
  52. data/lib/wurk/timer_loop.rb +49 -0
  53. data/lib/wurk/version.rb +1 -1
  54. data/lib/wurk/web/extension.rb +4 -1
  55. data/lib/wurk/web/pool_scope.rb +46 -0
  56. data/lib/wurk/web/rack_app.rb +2 -1
  57. data/lib/wurk/web/search.rb +77 -18
  58. data/lib/wurk/web.rb +1 -0
  59. data/lib/wurk/worker/setter.rb +6 -1
  60. data/lib/wurk.rb +6 -2
  61. data/vendor/assets/dashboard/assets/ArgsValue-DYfBiXrJ.js +1 -0
  62. data/vendor/assets/dashboard/assets/BatchDetail-DTZ2HzcD.js +1 -0
  63. data/vendor/assets/dashboard/assets/Batches-BrnXA332.js +1 -0
  64. data/vendor/assets/dashboard/assets/Busy-C57G8Xb3.js +1 -0
  65. data/vendor/assets/dashboard/assets/Cron-DlyH88oo.js +1 -0
  66. data/vendor/assets/dashboard/assets/Dashboard-B887pxWf.js +1 -0
  67. data/vendor/assets/dashboard/assets/Dead-BPA7gs-X.js +1 -0
  68. data/vendor/assets/dashboard/assets/Extension-Bunf6XuU.js +1 -0
  69. data/vendor/assets/dashboard/assets/FilterBox-FCDi4ZCU.js +1 -0
  70. data/vendor/assets/dashboard/assets/JobDetailModal-DuMdKUMm.js +2 -0
  71. data/vendor/assets/dashboard/assets/Limiters-Br0aCPMK.js +1 -0
  72. data/vendor/assets/dashboard/assets/Metrics-DxBmuywH.js +1 -0
  73. data/vendor/assets/dashboard/assets/Modal-t4FI_LaY.js +1 -0
  74. data/vendor/assets/dashboard/assets/PageHeader-CsDvJSOA.js +1 -0
  75. data/vendor/assets/dashboard/assets/Profiles-Bkhoqjlq.js +1 -0
  76. data/vendor/assets/dashboard/assets/Queues-BvhA-vfI.js +1 -0
  77. data/vendor/assets/dashboard/assets/Retries-JEpB-1Yl.js +1 -0
  78. data/vendor/assets/dashboard/assets/Scheduled-DN_FbSwP.js +1 -0
  79. data/vendor/assets/dashboard/assets/Search-DJuK0YCJ.js +1 -0
  80. data/vendor/assets/dashboard/assets/Skeleton-DOYDkzg1.js +1 -0
  81. data/vendor/assets/dashboard/assets/charts-CVK0zAnC.js +1 -0
  82. data/vendor/assets/dashboard/assets/index-BnPX9Ptn.css +1 -0
  83. data/vendor/assets/dashboard/assets/index-CZTcs-pM.js +141 -0
  84. data/vendor/assets/dashboard/assets/useResetPageOnEmpty-CoZU4b3a.js +1 -0
  85. data/vendor/assets/dashboard/assets/useSort-wQcnbdsa.js +1 -0
  86. data/vendor/assets/dashboard/index.html +3 -3
  87. data/vendor/assets/dashboard/wurk-manifest.json +2 -2
  88. metadata +55 -26
  89. data/vendor/assets/dashboard/assets/ArgsValue-BUqJa-eG.js +0 -1
  90. data/vendor/assets/dashboard/assets/BatchDetail-C5dkqAzp.js +0 -1
  91. data/vendor/assets/dashboard/assets/Batches-bgkXn8tc.js +0 -1
  92. data/vendor/assets/dashboard/assets/Busy-QSHBFXhy.js +0 -1
  93. data/vendor/assets/dashboard/assets/Cron-CUHppvTA.js +0 -1
  94. data/vendor/assets/dashboard/assets/Dashboard-CzPudckV.js +0 -1
  95. data/vendor/assets/dashboard/assets/Dead-BIq4Nz_i.js +0 -1
  96. data/vendor/assets/dashboard/assets/Extension-CW36i9R1.js +0 -1
  97. data/vendor/assets/dashboard/assets/JobDetailModal-BhNdWSp7.js +0 -2
  98. data/vendor/assets/dashboard/assets/Limiters-CiI_DfUK.js +0 -1
  99. data/vendor/assets/dashboard/assets/Metrics-BzZ8ugms.js +0 -1
  100. data/vendor/assets/dashboard/assets/Modal-DzVfgsSF.js +0 -1
  101. data/vendor/assets/dashboard/assets/PageHeader-Dp3qhX3e.js +0 -1
  102. data/vendor/assets/dashboard/assets/Profiles-BTtIlTdR.js +0 -1
  103. data/vendor/assets/dashboard/assets/Queues-BuXoxQ4W.js +0 -1
  104. data/vendor/assets/dashboard/assets/Retries-D4HAPaOQ.js +0 -1
  105. data/vendor/assets/dashboard/assets/Scheduled-6NCZYVJh.js +0 -1
  106. data/vendor/assets/dashboard/assets/Search-JaB_-52c.js +0 -1
  107. data/vendor/assets/dashboard/assets/charts-6uvCyY0x.js +0 -1
  108. data/vendor/assets/dashboard/assets/i18n-gIeA5VLo.js +0 -1
  109. data/vendor/assets/dashboard/assets/index-BDG9tvBA.css +0 -1
  110. data/vendor/assets/dashboard/assets/index-DWfWAGBc.js +0 -141
  111. data/vendor/assets/dashboard/assets/useJobSetActions-DTDaAoWr.js +0 -1
  112. data/vendor/assets/dashboard/assets/usePageParam-BcAvRko-.js +0 -1
  113. data/vendor/assets/dashboard/assets/useSort-D5Am4bGq.js +0 -1
data/lib/wurk/manager.rb CHANGED
@@ -38,15 +38,25 @@ module Wurk
38
38
  end
39
39
 
40
40
  def start
41
- @workers.each(&:start)
41
+ workers_snapshot.each(&:start)
42
42
  end
43
43
 
44
44
  def quiet
45
45
  return if @done
46
46
 
47
- @done = true
47
+ snapshot = @plock.synchronize do
48
+ @done = true
49
+ @workers.dup
50
+ end
51
+ # Halt fetching for the whole capsule up front: the shared fetcher's drain
52
+ # flag makes retrieve_work return nil immediately, so a processor can't pull
53
+ # a fresh job between quiet and its own terminate taking effect. Safe-nav
54
+ # covers a quiet that lands before Capsule#prepare! materializes the fetcher
55
+ # (e.g. a signal-driven quiet on a partially-booted launcher) — nothing is
56
+ # fetching yet, so there is nothing to halt.
57
+ capsule.fetcher&.terminate
48
58
  logger.info { "Terminating quiet threads for #{capsule.name} capsule" }
49
- @workers.each(&:terminate)
59
+ snapshot.each(&:terminate)
50
60
  end
51
61
 
52
62
  # Graceful shutdown: quiet first, then poll for workers to clear.
@@ -57,11 +67,11 @@ module Wurk
57
67
  # Lifecycle hooks (e.g. :quiet) can be async; give them a tick to settle
58
68
  # before we start polling. Matches Sidekiq's PAUSE_TIME behavior.
59
69
  sleep PAUSE_TIME
60
- return if @workers.empty?
70
+ return if workers_empty?
61
71
 
62
72
  logger.info { 'Pausing to allow jobs to finish...' }
63
- wait_for(deadline) { @workers.empty? }
64
- return if @workers.empty?
73
+ wait_for(deadline) { workers_empty? }
74
+ return if workers_empty?
65
75
 
66
76
  hard_shutdown
67
77
  ensure
@@ -75,27 +85,38 @@ module Wurk
75
85
  # Processor#run callback: invoked when a Processor thread exits, whether
76
86
  # cleanly or via raised exception. Removes the dead processor from the
77
87
  # pool and (unless we're already stopping) spawns a replacement so the
78
- # capsule's concurrency stays constant.
88
+ # capsule's concurrency stays constant. If the replacement itself can't be
89
+ # spawned, crash the child (the swarm respawns it) rather than silently
90
+ # dropping concurrency. Snapshot under @plock; start the replacement — a
91
+ # side effect — outside the lock.
79
92
  def processor_result(processor, _reason = nil)
80
- @plock.synchronize do
93
+ replacement = @plock.synchronize do
81
94
  @workers.delete(processor)
82
95
  unless @done
83
96
  p = Processor.new(@capsule, &method(:processor_result))
84
97
  @workers << p
85
- p.start
98
+ p
86
99
  end
87
100
  end
101
+ replacement&.start
102
+ rescue StandardError => e
103
+ # Replacement spawn failed (e.g. ThreadError at the OS thread limit).
104
+ # Silently running one Processor short for the life of the process is
105
+ # invisible degradation; instead report and crash the child on the main
106
+ # thread so the swarm respawns it at full concurrency (plan 02 §6).
107
+ @capsule.config.handle_exception(e, { context: 'Manager could not replace a dead Processor' })
108
+ main_thread.raise(e)
88
109
  end
89
110
 
90
- # Reached when the deadline expired with workers still busy. We must
91
- # push their in-flight UoWs back to the public queues BEFORE raising
92
- # Wurk::Shutdown into the threads — losing a job is worse than running
93
- # it twice (Sidekiq's at-least-once contract).
111
+ # Reached when the deadline expired with workers still busy. Atomically
112
+ # move their in-flight UoWs private→public (Reliable#bulk_requeue) BEFORE
113
+ # raising Wurk::Shutdown into the threads, so a job killed mid-perform is
114
+ # re-run once (Sidekiq's at-least-once contract). `job` is read off another
115
+ # thread, so a Processor can ACK between this map and the requeue — but
116
+ # bulk_requeue's LREM guard skips the RPUSH on a miss, so a job that
117
+ # finished in that window is not resurrected onto the public queue.
94
118
  def hard_shutdown # rubocop:disable Metrics/AbcSize
95
- cleanup = nil
96
- @plock.synchronize do
97
- cleanup = @workers.dup
98
- end
119
+ cleanup = workers_snapshot
99
120
 
100
121
  if cleanup.any?
101
122
  jobs = cleanup.map(&:job).compact
@@ -103,7 +124,9 @@ module Wurk
103
124
  logger.warn { "Terminating #{cleanup.size} busy threads" }
104
125
  logger.debug { "Jobs still in progress #{jobs.inspect}" }
105
126
 
106
- capsule.fetcher.bulk_requeue(jobs)
127
+ # `&.` like #quiet: a TERM in the pre-prepare! window (traps install
128
+ # before launcher.run) reaches here with no fetcher built yet.
129
+ capsule.fetcher&.bulk_requeue(jobs)
107
130
  end
108
131
 
109
132
  cleanup.each(&:kill)
@@ -111,11 +134,28 @@ module Wurk
111
134
  # The caller typically `exit`s immediately after we return; give
112
135
  # threads a brief window to run their `ensure` blocks.
113
136
  deadline = ::Process.clock_gettime(::Process::CLOCK_MONOTONIC) + 3
114
- wait_for(deadline) { @workers.empty? }
137
+ wait_for(deadline) { workers_empty? }
115
138
  end
116
139
 
117
140
  private
118
141
 
142
+ # The @workers Set is mutated from Processor threads (processor_result)
143
+ # while the lifecycle methods read/iterate it from the manager thread.
144
+ # Snapshot under @plock, then act on the copy outside the lock.
145
+ def workers_snapshot
146
+ @plock.synchronize { @workers.dup }
147
+ end
148
+
149
+ def workers_empty?
150
+ @plock.synchronize { @workers.empty? }
151
+ end
152
+
153
+ # Seam over Thread.main: lets processor_result's replacement-failure crash
154
+ # be unit-tested against a controlled thread instead of the live runner.
155
+ def main_thread
156
+ Thread.main
157
+ end
158
+
119
159
  # Polls `condblock` until it returns true or the monotonic deadline
120
160
  # passes. The PAUSE_TIME floor stops us from spinning when only a few
121
161
  # milliseconds remain.
@@ -6,42 +6,42 @@ module Wurk
6
6
  module Metrics
7
7
  # Ent feature parity (§5): server middleware that records per-job-class
8
8
  # execution metrics into Redis time-buckets. The on-the-wire schema is
9
- # wire-compat with Sidekiq 8.x's history pane so dashboards built against
10
- # `j|YYMMDD|H:M` HASH keys keep working unchanged.
9
+ # wire-compat with Sidekiq's history pane — Sidekiq keys the per-minute
10
+ # HASH as `j|<YYYYMMDD>|<H>:<M>`, so dashboards (and Sidekiq data migrated
11
+ # in place) keep resolving against the same key after a drop-in swap.
11
12
  #
12
13
  # Bucket layout (spec: docs/target/sidekiq-free.md §1.6):
13
14
  #
14
- # j|YYMMDD|H:M HASH per-minute bucket, TTL = MID_TERM (3 days)
15
- # <klass>|p INT processed count
16
- # <klass>|f INT failed count
17
- # <klass>|ms INT total ms spent
15
+ # j|YYYYMMDD|H:M HASH per-minute bucket, TTL = MID_TERM (3 days)
16
+ # <klass>|p INT processed count
17
+ # <klass>|f INT failed count
18
+ # <klass>|ms INT total ms spent
18
19
  #
19
- # j|YYMMDD|H:m0 HASH 10-minute rollup (last digit zeroed),
20
- # TTL = SHORT_TERM (8 hours) — short window for
21
- # quick aggregate queries without scanning 600 minute keys.
20
+ # <klass>-YYYYMMDD-H HASH per-class hourly histogram, TTL = MID_TERM
22
21
  #
23
- # <klass>-YYMMDD-H HASH per-class hourly histogram, TTL = MID_TERM
22
+ # We deliberately do NOT write a `H:m0` 10-minute rollup. Its key format
23
+ # collides with the real minute-0 bucket, so rolling x1..x9 into it turns
24
+ # that minute's value into a decade total — and the read side (Query) then
25
+ # sums the minute-0 bucket alongside x1..x9 and double-counts. Sidekiq
26
+ # itself doesn't keep that rollup (the daily/hourly rollups are commented
27
+ # out in its ExecutionTracker); Query reads the last N per-minute keys.
24
28
  #
25
- # Every bucket TTL is set on first write (EXPIRE NX-equivalent: only when
26
- # the HASH was newly created in this call) — re-asserting TTL on every
27
- # write would keep the bucket alive indefinitely while traffic continues,
28
- # but that's the desired behavior here: as long as a class keeps running,
29
- # we keep the minute bucket around for the retention window measured from
29
+ # Every bucket TTL is set on every write (not NX): as long as a class keeps
30
+ # running we keep its bucket around for the retention window measured from
30
31
  # *last write*, not from first write. So we EXPIRE unconditionally.
31
32
  #
32
- # The middleware is hot-path — every successful job pays for it. Writes
33
- # are pipelined in a single round-trip per job (1 HINCRBY × 3 + 1 EXPIRE
34
- # per bucket × 3 buckets = 12 commands, batched).
33
+ # The middleware is hot-path — every successful job pays for it. Writes are
34
+ # pipelined in a single round-trip per job (1 HINCRBY × 2 + 1 EXPIRE per
35
+ # bucket × 2 buckets = 6 commands, batched).
35
36
  class History
36
37
  include Wurk::Middleware::ServerMiddleware
37
38
 
38
39
  # Per spec §1.6 — naming mirrors the upstream constants so anyone
39
40
  # grepping the Sidekiq source for `MID_TERM` lands here.
40
41
  MID_TERM = 3 * 24 * 60 * 60 # 3 days, in seconds
41
- SHORT_TERM = 8 * 60 * 60 # 8 hours, in seconds
42
42
 
43
43
  MINUTE_KEY_PREFIX = 'j|'
44
- DATE_FORMAT = '%y%m%d' # YYMMDD — two-digit year per spec
44
+ DATE_FORMAT = '%Y%m%d' # YYYYMMDD — matches Sidekiq's j| key
45
45
 
46
46
  def call(_worker, job, _queue)
47
47
  klass = job['class']
@@ -74,28 +74,20 @@ module Wurk
74
74
 
75
75
  ms = duration_ms.to_i
76
76
  ms = 0 if ms.negative?
77
- buckets = { minute: minute_key(at), rollup: rollup_key(at), hour: hour_key(klass, at) }
77
+ buckets = { minute: minute_key(at), hour: hour_key(klass, at) }
78
78
  with_pool(redis_pool) { |conn| pipeline_write(conn, klass, ms, success, buckets) }
79
79
  nil
80
80
  end
81
81
 
82
- # Minute + 10-min rollup share a per-class `|p|f|ms` field layout;
83
- # the hourly bucket is already class-scoped so its fields are bare
84
- # `p|f|ms`. Pipeline all 9 commands in one round-trip.
82
+ # The minute bucket carries per-class `<klass>|p|f|ms` fields; the
83
+ # hourly bucket is already class-scoped so its fields are bare
84
+ # `p|f|ms`. Pipeline all 6 commands in one round-trip.
85
85
  def pipeline_write(conn, klass, ms, success, buckets)
86
86
  outcome = success ? 'p' : 'f'
87
87
  class_outcome = "#{klass}|#{outcome}"
88
88
  class_ms = "#{klass}|ms"
89
89
  conn.pipelined do |pipe|
90
90
  incr_bucket(pipe, buckets[:minute], [class_outcome, class_ms, ms, MID_TERM])
91
- # The minute key and the 10-min rollup key coincide whenever the
92
- # minute ends in 0 (rollup zeroes the last digit). Writing both
93
- # would double-count the shared field — the minute write above
94
- # already lands on it — so skip the rollup write then. Minutes
95
- # x1..x9 still accumulate into the x0 rollup key as normal.
96
- unless buckets[:rollup] == buckets[:minute]
97
- incr_bucket(pipe, buckets[:rollup], [class_outcome, class_ms, ms, SHORT_TERM])
98
- end
99
91
  incr_bucket(pipe, buckets[:hour], [outcome, 'ms', ms, MID_TERM])
100
92
  end
101
93
  end
@@ -115,12 +107,6 @@ module Wurk
115
107
  date: t.strftime(DATE_FORMAT), hr: t.hour, min: t.min)
116
108
  end
117
109
 
118
- def rollup_key(time)
119
- t = time.utc
120
- format("#{MINUTE_KEY_PREFIX}%<date>s|%<hr>d:%<min>d",
121
- date: t.strftime(DATE_FORMAT), hr: t.hour, min: (t.min / 10) * 10)
122
- end
123
-
124
110
  def hour_key(klass, time)
125
111
  t = time.utc
126
112
  "#{klass}-#{t.strftime(DATE_FORMAT)}-#{t.hour}"
@@ -3,6 +3,7 @@
3
3
  require_relative '../component'
4
4
  require_relative '../keys'
5
5
  require_relative '../job_record'
6
+ require_relative '../timer_loop'
6
7
  require_relative 'rollup'
7
8
 
8
9
  module Wurk
@@ -45,28 +46,16 @@ module Wurk
45
46
 
46
47
  def initialize(config)
47
48
  @config = config
48
- @done = false
49
- @mutex = ::Mutex.new
50
- @sleeper = ::ConditionVariable.new
51
- @tick_interval = config[:metrics_rollup_interval] || DEFAULT_TICK_SECONDS
49
+ @timer = TimerLoop.new(config[:metrics_rollup_interval] || DEFAULT_TICK_SECONDS)
52
50
  @thread = nil
53
51
  end
54
52
 
55
53
  def start
56
- @thread ||= safe_thread('queue-metrics') do # rubocop:disable Naming/MemoizedInstanceVariableName
57
- wait
58
- until @done
59
- tick
60
- wait
61
- end
62
- end
54
+ @thread ||= safe_thread('queue-metrics') { @timer.run { tick } } # rubocop:disable Naming/MemoizedInstanceVariableName
63
55
  end
64
56
 
65
57
  def terminate
66
- @mutex.synchronize do
67
- @done = true
68
- @sleeper.signal
69
- end
58
+ @timer.terminate
70
59
  end
71
60
 
72
61
  # Leader-gated: only the elected leader samples, so N workers don't each
@@ -140,12 +129,6 @@ module Wurk
140
129
  rescue ::JSON::ParserError, ::TypeError, ::ArgumentError
141
130
  0.0
142
131
  end
143
-
144
- def wait
145
- @mutex.synchronize do
146
- @sleeper.wait(@mutex, @tick_interval) unless @done
147
- end
148
- end
149
132
  end
150
133
  end
151
134
  end
@@ -1,6 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  require_relative '../component'
4
+ require_relative '../timer_loop'
4
5
  require_relative 'history'
5
6
 
6
7
  module Wurk
@@ -53,28 +54,16 @@ module Wurk
53
54
 
54
55
  def initialize(config)
55
56
  @config = config
56
- @done = false
57
- @mutex = ::Mutex.new
58
- @sleeper = ::ConditionVariable.new
59
- @tick_interval = config[:metrics_rollup_interval] || DEFAULT_TICK_SECONDS
57
+ @timer = TimerLoop.new(config[:metrics_rollup_interval] || DEFAULT_TICK_SECONDS)
60
58
  @thread = nil
61
59
  end
62
60
 
63
61
  def start
64
- @thread ||= safe_thread('metrics-rollup') do # rubocop:disable Naming/MemoizedInstanceVariableName
65
- wait
66
- until @done
67
- tick
68
- wait
69
- end
70
- end
62
+ @thread ||= safe_thread('metrics-rollup') { @timer.run { tick } } # rubocop:disable Naming/MemoizedInstanceVariableName
71
63
  end
72
64
 
73
65
  def terminate
74
- @mutex.synchronize do
75
- @done = true
76
- @sleeper.signal
77
- end
66
+ @timer.terminate
78
67
  end
79
68
 
80
69
  # Leader-gated: only the elected leader writes the cluster-total series,
@@ -158,12 +147,6 @@ module Wurk
158
147
  def floor_min(time)
159
148
  (time.to_i / 60) * 60
160
149
  end
161
-
162
- def wait
163
- @mutex.synchronize do
164
- @sleeper.wait(@mutex, @tick_interval) unless @done
165
- end
166
- end
167
150
  end
168
151
  end
169
152
  end
@@ -2,9 +2,9 @@
2
2
 
3
3
  require_relative 'component'
4
4
  require_relative 'context'
5
+ require_relative 'dead_set'
5
6
  require_relative 'job_logger'
6
7
  require_relative 'job_retry'
7
- require_relative 'keys'
8
8
  require_relative 'profiler'
9
9
 
10
10
  module Wurk
@@ -175,8 +175,9 @@ module Wurk
175
175
  ack = true
176
176
  end
177
177
  rescue Wurk::JobRetry::Handled
178
- # JobRetry::Skip (subclass) or Handled — retry layer / middleware has
179
- # already booked the outcome; safe to ack.
178
+ # Handled / JobRetry::Skip (incl. Limiter::Rescheduled, where the
179
+ # limiter middleware already re-enqueued the job) — the retry layer or
180
+ # a middleware booked the outcome and recorded no retry; ack the UoW.
180
181
  ack = true
181
182
  rescue Wurk::Shutdown
182
183
  # Don't ack — UoW stays in private list and is reclaimed on reboot.
@@ -185,16 +186,15 @@ module Wurk
185
186
  end
186
187
  end
187
188
 
188
- # Parse JSON; on failure ZADD the raw payload to the dead set and ack.
189
+ # Parse JSON; on failure send the raw payload to the dead set (ZADD +
190
+ # trim, via the shared DeadSet#kill_raw so the malformed path caps the
191
+ # morgue exactly like send_to_morgue does — spec §31.8/§31.9) and ack.
189
192
  # Returns nil to signal "no further processing".
190
193
  def parse_or_kill(jobstr, uow)
191
194
  Wurk.load_json(jobstr)
192
195
  rescue ::JSON::ParserError => e
193
196
  handle_exception(e, { context: 'Invalid JSON', jobstr: jobstr })
194
- now = ::Process.clock_gettime(::Process::CLOCK_REALTIME)
195
- @capsule.redis do |conn|
196
- conn.call('ZADD', Keys::DEAD, now.to_s, jobstr)
197
- end
197
+ DeadSet.new.kill_raw(jobstr)
198
198
  uow.acknowledge
199
199
  nil
200
200
  end
@@ -21,16 +21,23 @@ module Wurk
21
21
 
22
22
  def size = @keys.size
23
23
 
24
+ # HMGET of the metadata fields only, pipelined into one round-trip:
25
+ # HGETALL would also pull each profile's `data` field — the multi-MB
26
+ # gzipped blob — through Redis for every list render.
27
+ METADATA_FIELDS = %w[jid type token size elapsed started_at].freeze
28
+
24
29
  def each
25
30
  return enum_for(:each) unless block_given?
26
31
 
27
- Wurk.redis do |conn|
28
- @keys.each do |key|
29
- raw = conn.call('HGETALL', key)
30
- hash = raw.is_a?(Hash) ? raw : raw.each_slice(2).to_h
31
- yield ProfileRecord.new(hash) unless hash.empty?
32
+ rows = Wurk.redis do |conn|
33
+ conn.pipelined do |pipe|
34
+ @keys.each { |key| pipe.call('HMGET', key, *METADATA_FIELDS) }
32
35
  end
33
36
  end
37
+ rows.each do |values|
38
+ hash = METADATA_FIELDS.zip(Array(values)).to_h
39
+ yield ProfileRecord.new(hash) unless hash['jid'].nil?
40
+ end
34
41
  end
35
42
  end
36
43
 
@@ -39,6 +46,14 @@ module Wurk
39
46
  class ProfileRecord
40
47
  attr_reader :jid, :type, :token, :size, :elapsed
41
48
 
49
+ # Fetch the stored gzipped blob for a profile storage key ("<token>-<jid>")
50
+ # straight from Redis, without materializing the whole record — the Profiles
51
+ # data endpoint streams it to the browser as-is. nil if the HASH is gone.
52
+ # Owns the `data` HASH-field name so web callers don't hardcode the schema.
53
+ def self.data_for(key)
54
+ Wurk.redis { |conn| conn.call('HGET', key, 'data') }
55
+ end
56
+
42
57
  def initialize(hash)
43
58
  @hash = hash
44
59
  @jid = hash['jid']
@@ -59,7 +74,7 @@ module Wurk
59
74
  # bytes — the web layer streams them straight to the browser with a gzip
60
75
  # Content-Encoding. Returns nil if the HASH expired between list and read.
61
76
  def data
62
- Wurk.redis { |conn| conn.call('HGET', key, 'data') }
77
+ self.class.data_for(key)
63
78
  end
64
79
  end
65
80
  end
data/lib/wurk/profiler.rb CHANGED
@@ -50,11 +50,13 @@ module Wurk
50
50
  key = profile_key(token, jid)
51
51
  gz = gzip(gecko_json)
52
52
  with_pool(pool) do |conn|
53
- conn.call('HSET', key, 'jid', jid, 'type', type, 'token', token,
54
- 'started_at', started_at.to_i, 'elapsed', elapsed_ms.to_i,
55
- 'size', gz.bytesize, 'sid', sid.to_s, 'data', gz)
56
- conn.call('EXPIRE', key, TTL)
57
- conn.call('ZADD', Keys::PROFILES, (now + TTL).to_i, key)
53
+ conn.pipelined do |pipe|
54
+ pipe.call('HSET', key, 'jid', jid, 'type', type, 'token', token,
55
+ 'started_at', started_at.to_i, 'elapsed', elapsed_ms.to_i,
56
+ 'size', gz.bytesize, 'sid', sid.to_s, 'data', gz)
57
+ pipe.call('EXPIRE', key, TTL)
58
+ pipe.call('ZADD', Keys::PROFILES, (now + TTL).to_i, key)
59
+ end
58
60
  end
59
61
  key
60
62
  end
@@ -0,0 +1,176 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Wurk
4
+ # Coordinates how Wurk boots inside a Rails host: whether this process should
5
+ # boot at all, whether it may fork the swarm, and the boot itself. The Railtie
6
+ # owns only the Rails hooks (config namespace + server_mode initializer +
7
+ # after_initialize) and delegates every decision and side effect here, so the
8
+ # boot policy stays pure and unit-testable without the Railtie DSL.
9
+ # See docs/idea/03-process-model.md for the exact ordering.
10
+ module RailsBoot
11
+ module_function
12
+
13
+ # Invoked from the `wurk.server_mode` initializer, before config/initializers
14
+ # load. This Rails process forks the workers, so it IS the server. Enter
15
+ # server mode now — otherwise the app's `Sidekiq.configure_server` blocks
16
+ # gate on `config.server?` (still false) and are silently dropped. A process
17
+ # that won't run workers (skip_boot?, or one that refuses to boot under a
18
+ # preforking web server) is not a server.
19
+ def enter_server_mode_if_serving(app = ::Rails.application)
20
+ return if skip_boot?
21
+ return if boot_action(app) == :refuse
22
+
23
+ Wurk.enter_server_mode
24
+ end
25
+
26
+ # Invoked from after_initialize, once the host app has fully initialized.
27
+ def boot(app = ::Rails.application)
28
+ return if skip_boot?
29
+
30
+ case boot_action(app)
31
+ when :fork then boot_swarm
32
+ when :embed then boot_embedded
33
+ when :refuse then refuse_preforking_boot
34
+ end
35
+ end
36
+
37
+ # A process that won't run workers isn't a server: skip both server mode
38
+ # and the swarm boot. Console mode is detected reliably here — the console
39
+ # command file defines ::Rails::Console before initializers run.
40
+ def skip_boot?
41
+ ENV['WURK_DISABLED'] == '1' ||
42
+ building? ||
43
+ defined?(::Rails::Console) ||
44
+ ::Rails.env.test?
45
+ end
46
+
47
+ # What boot should do once skip_boot? is false. Pure — reads env / loaded
48
+ # constants / host config, no side effects — so the boot decision is
49
+ # unit-testable without forking:
50
+ # :fork — safe to fork the swarm in the background (the default).
51
+ # :refuse — a preforking web server owns process forking here; don't.
52
+ # :embed — host opted into in-process threads-only via embed_in_web.
53
+ def boot_action(app = ::Rails.application)
54
+ return :fork unless preforking_web_server?
55
+
56
+ embed_in_web?(app) ? :embed : :refuse
57
+ end
58
+
59
+ # Preforking / clustered web servers (Puma cluster, Unicorn, Passenger)
60
+ # fork their own worker processes. Forking the swarm from `after_initialize`
61
+ # in one of them is the highest-risk boot path: without app preloading every
62
+ # server-worker re-runs the hook and forks its own full swarm (N×
63
+ # oversubscription); with preloading the swarm supervisor ends up entangled
64
+ # with the server's own fork/signal supervision. Detect the common three.
65
+ def preforking_web_server?
66
+ return true if defined?(::PhusionPassenger)
67
+ return true if defined?(::Unicorn)
68
+
69
+ puma_cluster?
70
+ end
71
+
72
+ # Puma only preforks in cluster mode (workers > 0); single mode is threaded
73
+ # and safe to co-host the swarm. Best-effort: a missed cluster falls through
74
+ # to the historical fork path; an over-eager match is escapable via
75
+ # embed_in_web / WURK_DISABLED / running the swarm as its own process.
76
+ def puma_cluster?
77
+ return false unless defined?(::Puma)
78
+
79
+ count = puma_worker_count
80
+ count.is_a?(Integer) && count.positive?
81
+ end
82
+
83
+ # Worker count from Puma's parsed CLI config when the server booted it, else
84
+ # from WEB_CONCURRENCY (the near-universal convention for Puma workers).
85
+ # Guarded — the accessor and options shape vary across Puma versions.
86
+ def puma_worker_count
87
+ cfg = ::Puma.cli_config if ::Puma.respond_to?(:cli_config)
88
+ workers = cfg&.options&.[](:workers)
89
+ return workers.to_i if workers
90
+
91
+ ENV['WEB_CONCURRENCY']&.to_i
92
+ rescue StandardError
93
+ nil
94
+ end
95
+
96
+ # Host opt-in (config/application.rb): run workers as in-process threads
97
+ # instead of forking, like Sidekiq embedded. Set it in application.rb, not
98
+ # an initializer — server mode is decided before initializers load.
99
+ def embed_in_web?(app = ::Rails.application)
100
+ app.config.wurk&.embed_in_web == true
101
+ rescue StandardError
102
+ false
103
+ end
104
+
105
+ def boot_swarm
106
+ swarm = Wurk::Swarm.new(topology: Wurk.configuration.topology,
107
+ shutdown_timeout: Wurk.configuration[:timeout] || Swarm::DEFAULT_SHUTDOWN_TIMEOUT)
108
+ # Co-hosted in the web process (e.g. Puma single mode): the host owns the
109
+ # process-wide TERM/INT traps. Installing the swarm's own would hijack
110
+ # them — a deploy TERM would drain the swarm but never stop the HTTP
111
+ # server. Let the host keep signal ownership and drain the swarm on its
112
+ # graceful exit (same contract as boot_embedded).
113
+ swarm.boot(install_signals: false)
114
+ at_exit { swarm.shutdown }
115
+ # supervise must still run somewhere or crashed children never respawn and
116
+ # memory checks never fire. A background thread keeps the host's main
117
+ # thread free to serve HTTP.
118
+ Thread.new do
119
+ swarm.supervise
120
+ rescue StandardError => e
121
+ logger.error { "wurk supervisor thread died: #{e.class}: #{e.message}" }
122
+ end
123
+ end
124
+
125
+ # Sidekiq-embedded parity: a threads-only worker inside the web process, no
126
+ # fork. Redis validation failure keeps the host serving HTTP (log + carry
127
+ # on). at_exit drains on the graceful shutdown the web server runs on TERM.
128
+ def boot_embedded
129
+ instance = Wurk::Embedded.new(Wurk.configuration)
130
+ instance.run
131
+ at_exit { instance.stop }
132
+ logger.info { 'wurk: running embedded in the web process (config.wurk.embed_in_web) — threads only, no fork' }
133
+ instance
134
+ rescue StandardError => e
135
+ logger.error { "wurk: embedded boot failed: #{e.class}: #{e.message}" }
136
+ nil
137
+ end
138
+
139
+ def refuse_preforking_boot
140
+ logger.warn { <<~MSG }
141
+ wurk: preforking web server detected (Puma cluster / Unicorn / Passenger).
142
+ Refusing to fork the worker swarm from a process that forks its own workers.
143
+ Run the swarm as its own process instead:
144
+
145
+ bundle exec wurkswarm # forked swarm, real parallelism
146
+ bundle exec wurk # single process, thread pool
147
+
148
+ Or run workers inside this web process (threads only, no fork):
149
+
150
+ # config/application.rb
151
+ config.wurk.embed_in_web = true
152
+
153
+ Already running workers elsewhere? Set WURK_DISABLED=1 here to silence this.
154
+ MSG
155
+ end
156
+
157
+ def logger
158
+ Wurk.configuration.logger
159
+ end
160
+
161
+ # A build/precompile step must never fork the swarm (#247). The default
162
+ # Rails Dockerfile runs `SECRET_KEY_BASE_DUMMY=1 ./bin/rails
163
+ # assets:precompile`; that loads `:environment` → fires after_initialize,
164
+ # but there's no Redis during `docker build`, so a fork would hang/fail the
165
+ # build. Same for other env-loading rake tasks (db:prepare, db:migrate).
166
+ # The real server path is unaffected: `rails server` / `puma` boot through
167
+ # Rails::Command, not Rake, and don't set the dummy secret.
168
+ def building?
169
+ return true if ENV.key?('SECRET_KEY_BASE_DUMMY')
170
+
171
+ defined?(::Rake) && ::Rake.application.top_level_tasks.any?
172
+ rescue StandardError
173
+ false
174
+ end
175
+ end
176
+ end