wurk 1.1.0 → 1.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +25 -0
- data/app/controllers/concerns/wurk/same_origin_guard.rb +40 -0
- data/app/controllers/concerns/wurk/sse_streaming.rb +48 -0
- data/app/controllers/concerns/wurk/stream_concurrency_guard.rb +53 -0
- data/app/controllers/wurk/api/pagination.rb +60 -11
- data/app/controllers/wurk/api/serializers.rb +5 -1
- data/app/controllers/wurk/api_controller.rb +51 -70
- data/app/controllers/wurk/application_controller.rb +26 -0
- data/app/controllers/wurk/dashboard_controller.rb +23 -1
- data/app/controllers/wurk/extensions_controller.rb +3 -12
- data/app/controllers/wurk/profiles_controller.rb +6 -1
- data/lib/wurk/batch/death_handler.rb +13 -0
- data/lib/wurk/batch/server_middleware.rb +9 -5
- data/lib/wurk/batch.rb +3 -0
- data/lib/wurk/capsule.rb +34 -21
- data/lib/wurk/cli.rb +16 -3
- data/lib/wurk/client/buffered.rb +30 -10
- data/lib/wurk/client.rb +45 -1
- data/lib/wurk/component.rb +23 -9
- data/lib/wurk/configuration.rb +83 -18
- data/lib/wurk/cron.rb +13 -1
- data/lib/wurk/dead_set.rb +16 -1
- data/lib/wurk/engine.rb +17 -1
- data/lib/wurk/fetcher/reliable.rb +46 -13
- data/lib/wurk/fetcher.rb +5 -0
- data/lib/wurk/health.rb +74 -22
- data/lib/wurk/history.rb +4 -21
- data/lib/wurk/job_retry.rb +3 -5
- data/lib/wurk/launcher.rb +26 -3
- data/lib/wurk/limiter/server_middleware.rb +16 -1
- data/lib/wurk/limiter.rb +6 -5
- data/lib/wurk/lua/loader.rb +11 -5
- data/lib/wurk/lua.rb +100 -9
- data/lib/wurk/manager.rb +59 -19
- data/lib/wurk/metrics/history.rb +24 -38
- data/lib/wurk/metrics/queue_rollup.rb +4 -21
- data/lib/wurk/metrics/rollup.rb +4 -21
- data/lib/wurk/processor.rb +8 -8
- data/lib/wurk/profile_set.rb +21 -6
- data/lib/wurk/profiler.rb +7 -5
- data/lib/wurk/rails_boot.rb +176 -0
- data/lib/wurk/railtie.rb +19 -47
- data/lib/wurk/redis_connection.rb +6 -9
- data/lib/wurk/redis_pool.rb +148 -28
- data/lib/wurk/scheduled.rb +35 -12
- data/lib/wurk/swarm/backoff.rb +70 -0
- data/lib/wurk/swarm/child_boot.rb +92 -13
- data/lib/wurk/swarm/orphan_guard.rb +105 -0
- data/lib/wurk/swarm/restart.rb +196 -0
- data/lib/wurk/swarm.rb +194 -78
- data/lib/wurk/timer_loop.rb +49 -0
- data/lib/wurk/version.rb +1 -1
- data/lib/wurk/web/extension.rb +4 -1
- data/lib/wurk/web/pool_scope.rb +46 -0
- data/lib/wurk/web/rack_app.rb +2 -1
- data/lib/wurk/web/search.rb +77 -18
- data/lib/wurk/web.rb +1 -0
- data/lib/wurk/worker/setter.rb +6 -1
- data/lib/wurk.rb +6 -2
- data/vendor/assets/dashboard/assets/ArgsValue-DYfBiXrJ.js +1 -0
- data/vendor/assets/dashboard/assets/BatchDetail-DTZ2HzcD.js +1 -0
- data/vendor/assets/dashboard/assets/Batches-BrnXA332.js +1 -0
- data/vendor/assets/dashboard/assets/Busy-C57G8Xb3.js +1 -0
- data/vendor/assets/dashboard/assets/Cron-DlyH88oo.js +1 -0
- data/vendor/assets/dashboard/assets/Dashboard-B887pxWf.js +1 -0
- data/vendor/assets/dashboard/assets/Dead-BPA7gs-X.js +1 -0
- data/vendor/assets/dashboard/assets/Extension-Bunf6XuU.js +1 -0
- data/vendor/assets/dashboard/assets/FilterBox-FCDi4ZCU.js +1 -0
- data/vendor/assets/dashboard/assets/JobDetailModal-DuMdKUMm.js +2 -0
- data/vendor/assets/dashboard/assets/Limiters-Br0aCPMK.js +1 -0
- data/vendor/assets/dashboard/assets/Metrics-DxBmuywH.js +1 -0
- data/vendor/assets/dashboard/assets/Modal-t4FI_LaY.js +1 -0
- data/vendor/assets/dashboard/assets/PageHeader-CsDvJSOA.js +1 -0
- data/vendor/assets/dashboard/assets/Profiles-Bkhoqjlq.js +1 -0
- data/vendor/assets/dashboard/assets/Queues-BvhA-vfI.js +1 -0
- data/vendor/assets/dashboard/assets/Retries-JEpB-1Yl.js +1 -0
- data/vendor/assets/dashboard/assets/Scheduled-DN_FbSwP.js +1 -0
- data/vendor/assets/dashboard/assets/Search-DJuK0YCJ.js +1 -0
- data/vendor/assets/dashboard/assets/Skeleton-DOYDkzg1.js +1 -0
- data/vendor/assets/dashboard/assets/charts-CVK0zAnC.js +1 -0
- data/vendor/assets/dashboard/assets/index-BnPX9Ptn.css +1 -0
- data/vendor/assets/dashboard/assets/index-CZTcs-pM.js +141 -0
- data/vendor/assets/dashboard/assets/useResetPageOnEmpty-CoZU4b3a.js +1 -0
- data/vendor/assets/dashboard/assets/useSort-wQcnbdsa.js +1 -0
- data/vendor/assets/dashboard/index.html +3 -3
- data/vendor/assets/dashboard/wurk-manifest.json +2 -2
- metadata +55 -26
- data/vendor/assets/dashboard/assets/ArgsValue-BUqJa-eG.js +0 -1
- data/vendor/assets/dashboard/assets/BatchDetail-C5dkqAzp.js +0 -1
- data/vendor/assets/dashboard/assets/Batches-bgkXn8tc.js +0 -1
- data/vendor/assets/dashboard/assets/Busy-QSHBFXhy.js +0 -1
- data/vendor/assets/dashboard/assets/Cron-CUHppvTA.js +0 -1
- data/vendor/assets/dashboard/assets/Dashboard-CzPudckV.js +0 -1
- data/vendor/assets/dashboard/assets/Dead-BIq4Nz_i.js +0 -1
- data/vendor/assets/dashboard/assets/Extension-CW36i9R1.js +0 -1
- data/vendor/assets/dashboard/assets/JobDetailModal-BhNdWSp7.js +0 -2
- data/vendor/assets/dashboard/assets/Limiters-CiI_DfUK.js +0 -1
- data/vendor/assets/dashboard/assets/Metrics-BzZ8ugms.js +0 -1
- data/vendor/assets/dashboard/assets/Modal-DzVfgsSF.js +0 -1
- data/vendor/assets/dashboard/assets/PageHeader-Dp3qhX3e.js +0 -1
- data/vendor/assets/dashboard/assets/Profiles-BTtIlTdR.js +0 -1
- data/vendor/assets/dashboard/assets/Queues-BuXoxQ4W.js +0 -1
- data/vendor/assets/dashboard/assets/Retries-D4HAPaOQ.js +0 -1
- data/vendor/assets/dashboard/assets/Scheduled-6NCZYVJh.js +0 -1
- data/vendor/assets/dashboard/assets/Search-JaB_-52c.js +0 -1
- data/vendor/assets/dashboard/assets/charts-6uvCyY0x.js +0 -1
- data/vendor/assets/dashboard/assets/i18n-gIeA5VLo.js +0 -1
- data/vendor/assets/dashboard/assets/index-BDG9tvBA.css +0 -1
- data/vendor/assets/dashboard/assets/index-DWfWAGBc.js +0 -141
- data/vendor/assets/dashboard/assets/useJobSetActions-DTDaAoWr.js +0 -1
- data/vendor/assets/dashboard/assets/usePageParam-BcAvRko-.js +0 -1
- data/vendor/assets/dashboard/assets/useSort-D5Am4bGq.js +0 -1
data/lib/wurk/manager.rb
CHANGED
|
@@ -38,15 +38,25 @@ module Wurk
|
|
|
38
38
|
end
|
|
39
39
|
|
|
40
40
|
def start
|
|
41
|
-
|
|
41
|
+
workers_snapshot.each(&:start)
|
|
42
42
|
end
|
|
43
43
|
|
|
44
44
|
def quiet
|
|
45
45
|
return if @done
|
|
46
46
|
|
|
47
|
-
|
|
47
|
+
snapshot = @plock.synchronize do
|
|
48
|
+
@done = true
|
|
49
|
+
@workers.dup
|
|
50
|
+
end
|
|
51
|
+
# Halt fetching for the whole capsule up front: the shared fetcher's drain
|
|
52
|
+
# flag makes retrieve_work return nil immediately, so a processor can't pull
|
|
53
|
+
# a fresh job between quiet and its own terminate taking effect. Safe-nav
|
|
54
|
+
# covers a quiet that lands before Capsule#prepare! materializes the fetcher
|
|
55
|
+
# (e.g. a signal-driven quiet on a partially-booted launcher) — nothing is
|
|
56
|
+
# fetching yet, so there is nothing to halt.
|
|
57
|
+
capsule.fetcher&.terminate
|
|
48
58
|
logger.info { "Terminating quiet threads for #{capsule.name} capsule" }
|
|
49
|
-
|
|
59
|
+
snapshot.each(&:terminate)
|
|
50
60
|
end
|
|
51
61
|
|
|
52
62
|
# Graceful shutdown: quiet first, then poll for workers to clear.
|
|
@@ -57,11 +67,11 @@ module Wurk
|
|
|
57
67
|
# Lifecycle hooks (e.g. :quiet) can be async; give them a tick to settle
|
|
58
68
|
# before we start polling. Matches Sidekiq's PAUSE_TIME behavior.
|
|
59
69
|
sleep PAUSE_TIME
|
|
60
|
-
return if
|
|
70
|
+
return if workers_empty?
|
|
61
71
|
|
|
62
72
|
logger.info { 'Pausing to allow jobs to finish...' }
|
|
63
|
-
wait_for(deadline) {
|
|
64
|
-
return if
|
|
73
|
+
wait_for(deadline) { workers_empty? }
|
|
74
|
+
return if workers_empty?
|
|
65
75
|
|
|
66
76
|
hard_shutdown
|
|
67
77
|
ensure
|
|
@@ -75,27 +85,38 @@ module Wurk
|
|
|
75
85
|
# Processor#run callback: invoked when a Processor thread exits, whether
|
|
76
86
|
# cleanly or via raised exception. Removes the dead processor from the
|
|
77
87
|
# pool and (unless we're already stopping) spawns a replacement so the
|
|
78
|
-
# capsule's concurrency stays constant.
|
|
88
|
+
# capsule's concurrency stays constant. If the replacement itself can't be
|
|
89
|
+
# spawned, crash the child (the swarm respawns it) rather than silently
|
|
90
|
+
# dropping concurrency. Snapshot under @plock; start the replacement — a
|
|
91
|
+
# side effect — outside the lock.
|
|
79
92
|
def processor_result(processor, _reason = nil)
|
|
80
|
-
@plock.synchronize do
|
|
93
|
+
replacement = @plock.synchronize do
|
|
81
94
|
@workers.delete(processor)
|
|
82
95
|
unless @done
|
|
83
96
|
p = Processor.new(@capsule, &method(:processor_result))
|
|
84
97
|
@workers << p
|
|
85
|
-
p
|
|
98
|
+
p
|
|
86
99
|
end
|
|
87
100
|
end
|
|
101
|
+
replacement&.start
|
|
102
|
+
rescue StandardError => e
|
|
103
|
+
# Replacement spawn failed (e.g. ThreadError at the OS thread limit).
|
|
104
|
+
# Silently running one Processor short for the life of the process is
|
|
105
|
+
# invisible degradation; instead report and crash the child on the main
|
|
106
|
+
# thread so the swarm respawns it at full concurrency (plan 02 §6).
|
|
107
|
+
@capsule.config.handle_exception(e, { context: 'Manager could not replace a dead Processor' })
|
|
108
|
+
main_thread.raise(e)
|
|
88
109
|
end
|
|
89
110
|
|
|
90
|
-
# Reached when the deadline expired with workers still busy.
|
|
91
|
-
#
|
|
92
|
-
# Wurk::Shutdown into the threads
|
|
93
|
-
#
|
|
111
|
+
# Reached when the deadline expired with workers still busy. Atomically
|
|
112
|
+
# move their in-flight UoWs private→public (Reliable#bulk_requeue) BEFORE
|
|
113
|
+
# raising Wurk::Shutdown into the threads, so a job killed mid-perform is
|
|
114
|
+
# re-run once (Sidekiq's at-least-once contract). `job` is read off another
|
|
115
|
+
# thread, so a Processor can ACK between this map and the requeue — but
|
|
116
|
+
# bulk_requeue's LREM guard skips the RPUSH on a miss, so a job that
|
|
117
|
+
# finished in that window is not resurrected onto the public queue.
|
|
94
118
|
def hard_shutdown # rubocop:disable Metrics/AbcSize
|
|
95
|
-
cleanup =
|
|
96
|
-
@plock.synchronize do
|
|
97
|
-
cleanup = @workers.dup
|
|
98
|
-
end
|
|
119
|
+
cleanup = workers_snapshot
|
|
99
120
|
|
|
100
121
|
if cleanup.any?
|
|
101
122
|
jobs = cleanup.map(&:job).compact
|
|
@@ -103,7 +124,9 @@ module Wurk
|
|
|
103
124
|
logger.warn { "Terminating #{cleanup.size} busy threads" }
|
|
104
125
|
logger.debug { "Jobs still in progress #{jobs.inspect}" }
|
|
105
126
|
|
|
106
|
-
|
|
127
|
+
# `&.` like #quiet: a TERM in the pre-prepare! window (traps install
|
|
128
|
+
# before launcher.run) reaches here with no fetcher built yet.
|
|
129
|
+
capsule.fetcher&.bulk_requeue(jobs)
|
|
107
130
|
end
|
|
108
131
|
|
|
109
132
|
cleanup.each(&:kill)
|
|
@@ -111,11 +134,28 @@ module Wurk
|
|
|
111
134
|
# The caller typically `exit`s immediately after we return; give
|
|
112
135
|
# threads a brief window to run their `ensure` blocks.
|
|
113
136
|
deadline = ::Process.clock_gettime(::Process::CLOCK_MONOTONIC) + 3
|
|
114
|
-
wait_for(deadline) {
|
|
137
|
+
wait_for(deadline) { workers_empty? }
|
|
115
138
|
end
|
|
116
139
|
|
|
117
140
|
private
|
|
118
141
|
|
|
142
|
+
# The @workers Set is mutated from Processor threads (processor_result)
|
|
143
|
+
# while the lifecycle methods read/iterate it from the manager thread.
|
|
144
|
+
# Snapshot under @plock, then act on the copy outside the lock.
|
|
145
|
+
def workers_snapshot
|
|
146
|
+
@plock.synchronize { @workers.dup }
|
|
147
|
+
end
|
|
148
|
+
|
|
149
|
+
def workers_empty?
|
|
150
|
+
@plock.synchronize { @workers.empty? }
|
|
151
|
+
end
|
|
152
|
+
|
|
153
|
+
# Seam over Thread.main: lets processor_result's replacement-failure crash
|
|
154
|
+
# be unit-tested against a controlled thread instead of the live runner.
|
|
155
|
+
def main_thread
|
|
156
|
+
Thread.main
|
|
157
|
+
end
|
|
158
|
+
|
|
119
159
|
# Polls `condblock` until it returns true or the monotonic deadline
|
|
120
160
|
# passes. The PAUSE_TIME floor stops us from spinning when only a few
|
|
121
161
|
# milliseconds remain.
|
data/lib/wurk/metrics/history.rb
CHANGED
|
@@ -6,42 +6,42 @@ module Wurk
|
|
|
6
6
|
module Metrics
|
|
7
7
|
# Ent feature parity (§5): server middleware that records per-job-class
|
|
8
8
|
# execution metrics into Redis time-buckets. The on-the-wire schema is
|
|
9
|
-
# wire-compat with Sidekiq
|
|
10
|
-
# `j
|
|
9
|
+
# wire-compat with Sidekiq's history pane — Sidekiq keys the per-minute
|
|
10
|
+
# HASH as `j|<YYYYMMDD>|<H>:<M>`, so dashboards (and Sidekiq data migrated
|
|
11
|
+
# in place) keep resolving against the same key after a drop-in swap.
|
|
11
12
|
#
|
|
12
13
|
# Bucket layout (spec: docs/target/sidekiq-free.md §1.6):
|
|
13
14
|
#
|
|
14
|
-
# j|
|
|
15
|
-
# <klass>|p
|
|
16
|
-
# <klass>|f
|
|
17
|
-
# <klass>|ms
|
|
15
|
+
# j|YYYYMMDD|H:M HASH per-minute bucket, TTL = MID_TERM (3 days)
|
|
16
|
+
# <klass>|p INT processed count
|
|
17
|
+
# <klass>|f INT failed count
|
|
18
|
+
# <klass>|ms INT total ms spent
|
|
18
19
|
#
|
|
19
|
-
#
|
|
20
|
-
# TTL = SHORT_TERM (8 hours) — short window for
|
|
21
|
-
# quick aggregate queries without scanning 600 minute keys.
|
|
20
|
+
# <klass>-YYYYMMDD-H HASH per-class hourly histogram, TTL = MID_TERM
|
|
22
21
|
#
|
|
23
|
-
#
|
|
22
|
+
# We deliberately do NOT write a `H:m0` 10-minute rollup. Its key format
|
|
23
|
+
# collides with the real minute-0 bucket, so rolling x1..x9 into it turns
|
|
24
|
+
# that minute's value into a decade total — and the read side (Query) then
|
|
25
|
+
# sums the minute-0 bucket alongside x1..x9 and double-counts. Sidekiq
|
|
26
|
+
# itself doesn't keep that rollup (the daily/hourly rollups are commented
|
|
27
|
+
# out in its ExecutionTracker); Query reads the last N per-minute keys.
|
|
24
28
|
#
|
|
25
|
-
# Every bucket TTL is set on
|
|
26
|
-
#
|
|
27
|
-
# write would keep the bucket alive indefinitely while traffic continues,
|
|
28
|
-
# but that's the desired behavior here: as long as a class keeps running,
|
|
29
|
-
# we keep the minute bucket around for the retention window measured from
|
|
29
|
+
# Every bucket TTL is set on every write (not NX): as long as a class keeps
|
|
30
|
+
# running we keep its bucket around for the retention window measured from
|
|
30
31
|
# *last write*, not from first write. So we EXPIRE unconditionally.
|
|
31
32
|
#
|
|
32
|
-
# The middleware is hot-path — every successful job pays for it. Writes
|
|
33
|
-
#
|
|
34
|
-
#
|
|
33
|
+
# The middleware is hot-path — every successful job pays for it. Writes are
|
|
34
|
+
# pipelined in a single round-trip per job (1 HINCRBY × 2 + 1 EXPIRE per
|
|
35
|
+
# bucket × 2 buckets = 6 commands, batched).
|
|
35
36
|
class History
|
|
36
37
|
include Wurk::Middleware::ServerMiddleware
|
|
37
38
|
|
|
38
39
|
# Per spec §1.6 — naming mirrors the upstream constants so anyone
|
|
39
40
|
# grepping the Sidekiq source for `MID_TERM` lands here.
|
|
40
41
|
MID_TERM = 3 * 24 * 60 * 60 # 3 days, in seconds
|
|
41
|
-
SHORT_TERM = 8 * 60 * 60 # 8 hours, in seconds
|
|
42
42
|
|
|
43
43
|
MINUTE_KEY_PREFIX = 'j|'
|
|
44
|
-
DATE_FORMAT = '%
|
|
44
|
+
DATE_FORMAT = '%Y%m%d' # YYYYMMDD — matches Sidekiq's j| key
|
|
45
45
|
|
|
46
46
|
def call(_worker, job, _queue)
|
|
47
47
|
klass = job['class']
|
|
@@ -74,28 +74,20 @@ module Wurk
|
|
|
74
74
|
|
|
75
75
|
ms = duration_ms.to_i
|
|
76
76
|
ms = 0 if ms.negative?
|
|
77
|
-
buckets = { minute: minute_key(at),
|
|
77
|
+
buckets = { minute: minute_key(at), hour: hour_key(klass, at) }
|
|
78
78
|
with_pool(redis_pool) { |conn| pipeline_write(conn, klass, ms, success, buckets) }
|
|
79
79
|
nil
|
|
80
80
|
end
|
|
81
81
|
|
|
82
|
-
#
|
|
83
|
-
#
|
|
84
|
-
# `p|f|ms`. Pipeline all
|
|
82
|
+
# The minute bucket carries per-class `<klass>|p|f|ms` fields; the
|
|
83
|
+
# hourly bucket is already class-scoped so its fields are bare
|
|
84
|
+
# `p|f|ms`. Pipeline all 6 commands in one round-trip.
|
|
85
85
|
def pipeline_write(conn, klass, ms, success, buckets)
|
|
86
86
|
outcome = success ? 'p' : 'f'
|
|
87
87
|
class_outcome = "#{klass}|#{outcome}"
|
|
88
88
|
class_ms = "#{klass}|ms"
|
|
89
89
|
conn.pipelined do |pipe|
|
|
90
90
|
incr_bucket(pipe, buckets[:minute], [class_outcome, class_ms, ms, MID_TERM])
|
|
91
|
-
# The minute key and the 10-min rollup key coincide whenever the
|
|
92
|
-
# minute ends in 0 (rollup zeroes the last digit). Writing both
|
|
93
|
-
# would double-count the shared field — the minute write above
|
|
94
|
-
# already lands on it — so skip the rollup write then. Minutes
|
|
95
|
-
# x1..x9 still accumulate into the x0 rollup key as normal.
|
|
96
|
-
unless buckets[:rollup] == buckets[:minute]
|
|
97
|
-
incr_bucket(pipe, buckets[:rollup], [class_outcome, class_ms, ms, SHORT_TERM])
|
|
98
|
-
end
|
|
99
91
|
incr_bucket(pipe, buckets[:hour], [outcome, 'ms', ms, MID_TERM])
|
|
100
92
|
end
|
|
101
93
|
end
|
|
@@ -115,12 +107,6 @@ module Wurk
|
|
|
115
107
|
date: t.strftime(DATE_FORMAT), hr: t.hour, min: t.min)
|
|
116
108
|
end
|
|
117
109
|
|
|
118
|
-
def rollup_key(time)
|
|
119
|
-
t = time.utc
|
|
120
|
-
format("#{MINUTE_KEY_PREFIX}%<date>s|%<hr>d:%<min>d",
|
|
121
|
-
date: t.strftime(DATE_FORMAT), hr: t.hour, min: (t.min / 10) * 10)
|
|
122
|
-
end
|
|
123
|
-
|
|
124
110
|
def hour_key(klass, time)
|
|
125
111
|
t = time.utc
|
|
126
112
|
"#{klass}-#{t.strftime(DATE_FORMAT)}-#{t.hour}"
|
|
@@ -3,6 +3,7 @@
|
|
|
3
3
|
require_relative '../component'
|
|
4
4
|
require_relative '../keys'
|
|
5
5
|
require_relative '../job_record'
|
|
6
|
+
require_relative '../timer_loop'
|
|
6
7
|
require_relative 'rollup'
|
|
7
8
|
|
|
8
9
|
module Wurk
|
|
@@ -45,28 +46,16 @@ module Wurk
|
|
|
45
46
|
|
|
46
47
|
def initialize(config)
|
|
47
48
|
@config = config
|
|
48
|
-
@
|
|
49
|
-
@mutex = ::Mutex.new
|
|
50
|
-
@sleeper = ::ConditionVariable.new
|
|
51
|
-
@tick_interval = config[:metrics_rollup_interval] || DEFAULT_TICK_SECONDS
|
|
49
|
+
@timer = TimerLoop.new(config[:metrics_rollup_interval] || DEFAULT_TICK_SECONDS)
|
|
52
50
|
@thread = nil
|
|
53
51
|
end
|
|
54
52
|
|
|
55
53
|
def start
|
|
56
|
-
@thread ||= safe_thread('queue-metrics')
|
|
57
|
-
wait
|
|
58
|
-
until @done
|
|
59
|
-
tick
|
|
60
|
-
wait
|
|
61
|
-
end
|
|
62
|
-
end
|
|
54
|
+
@thread ||= safe_thread('queue-metrics') { @timer.run { tick } } # rubocop:disable Naming/MemoizedInstanceVariableName
|
|
63
55
|
end
|
|
64
56
|
|
|
65
57
|
def terminate
|
|
66
|
-
@
|
|
67
|
-
@done = true
|
|
68
|
-
@sleeper.signal
|
|
69
|
-
end
|
|
58
|
+
@timer.terminate
|
|
70
59
|
end
|
|
71
60
|
|
|
72
61
|
# Leader-gated: only the elected leader samples, so N workers don't each
|
|
@@ -140,12 +129,6 @@ module Wurk
|
|
|
140
129
|
rescue ::JSON::ParserError, ::TypeError, ::ArgumentError
|
|
141
130
|
0.0
|
|
142
131
|
end
|
|
143
|
-
|
|
144
|
-
def wait
|
|
145
|
-
@mutex.synchronize do
|
|
146
|
-
@sleeper.wait(@mutex, @tick_interval) unless @done
|
|
147
|
-
end
|
|
148
|
-
end
|
|
149
132
|
end
|
|
150
133
|
end
|
|
151
134
|
end
|
data/lib/wurk/metrics/rollup.rb
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
require_relative '../component'
|
|
4
|
+
require_relative '../timer_loop'
|
|
4
5
|
require_relative 'history'
|
|
5
6
|
|
|
6
7
|
module Wurk
|
|
@@ -53,28 +54,16 @@ module Wurk
|
|
|
53
54
|
|
|
54
55
|
def initialize(config)
|
|
55
56
|
@config = config
|
|
56
|
-
@
|
|
57
|
-
@mutex = ::Mutex.new
|
|
58
|
-
@sleeper = ::ConditionVariable.new
|
|
59
|
-
@tick_interval = config[:metrics_rollup_interval] || DEFAULT_TICK_SECONDS
|
|
57
|
+
@timer = TimerLoop.new(config[:metrics_rollup_interval] || DEFAULT_TICK_SECONDS)
|
|
60
58
|
@thread = nil
|
|
61
59
|
end
|
|
62
60
|
|
|
63
61
|
def start
|
|
64
|
-
@thread ||= safe_thread('metrics-rollup')
|
|
65
|
-
wait
|
|
66
|
-
until @done
|
|
67
|
-
tick
|
|
68
|
-
wait
|
|
69
|
-
end
|
|
70
|
-
end
|
|
62
|
+
@thread ||= safe_thread('metrics-rollup') { @timer.run { tick } } # rubocop:disable Naming/MemoizedInstanceVariableName
|
|
71
63
|
end
|
|
72
64
|
|
|
73
65
|
def terminate
|
|
74
|
-
@
|
|
75
|
-
@done = true
|
|
76
|
-
@sleeper.signal
|
|
77
|
-
end
|
|
66
|
+
@timer.terminate
|
|
78
67
|
end
|
|
79
68
|
|
|
80
69
|
# Leader-gated: only the elected leader writes the cluster-total series,
|
|
@@ -158,12 +147,6 @@ module Wurk
|
|
|
158
147
|
def floor_min(time)
|
|
159
148
|
(time.to_i / 60) * 60
|
|
160
149
|
end
|
|
161
|
-
|
|
162
|
-
def wait
|
|
163
|
-
@mutex.synchronize do
|
|
164
|
-
@sleeper.wait(@mutex, @tick_interval) unless @done
|
|
165
|
-
end
|
|
166
|
-
end
|
|
167
150
|
end
|
|
168
151
|
end
|
|
169
152
|
end
|
data/lib/wurk/processor.rb
CHANGED
|
@@ -2,9 +2,9 @@
|
|
|
2
2
|
|
|
3
3
|
require_relative 'component'
|
|
4
4
|
require_relative 'context'
|
|
5
|
+
require_relative 'dead_set'
|
|
5
6
|
require_relative 'job_logger'
|
|
6
7
|
require_relative 'job_retry'
|
|
7
|
-
require_relative 'keys'
|
|
8
8
|
require_relative 'profiler'
|
|
9
9
|
|
|
10
10
|
module Wurk
|
|
@@ -175,8 +175,9 @@ module Wurk
|
|
|
175
175
|
ack = true
|
|
176
176
|
end
|
|
177
177
|
rescue Wurk::JobRetry::Handled
|
|
178
|
-
# JobRetry::Skip (
|
|
179
|
-
# already
|
|
178
|
+
# Handled / JobRetry::Skip (incl. Limiter::Rescheduled, where the
|
|
179
|
+
# limiter middleware already re-enqueued the job) — the retry layer or
|
|
180
|
+
# a middleware booked the outcome and recorded no retry; ack the UoW.
|
|
180
181
|
ack = true
|
|
181
182
|
rescue Wurk::Shutdown
|
|
182
183
|
# Don't ack — UoW stays in private list and is reclaimed on reboot.
|
|
@@ -185,16 +186,15 @@ module Wurk
|
|
|
185
186
|
end
|
|
186
187
|
end
|
|
187
188
|
|
|
188
|
-
# Parse JSON; on failure
|
|
189
|
+
# Parse JSON; on failure send the raw payload to the dead set (ZADD +
|
|
190
|
+
# trim, via the shared DeadSet#kill_raw so the malformed path caps the
|
|
191
|
+
# morgue exactly like send_to_morgue does — spec §31.8/§31.9) and ack.
|
|
189
192
|
# Returns nil to signal "no further processing".
|
|
190
193
|
def parse_or_kill(jobstr, uow)
|
|
191
194
|
Wurk.load_json(jobstr)
|
|
192
195
|
rescue ::JSON::ParserError => e
|
|
193
196
|
handle_exception(e, { context: 'Invalid JSON', jobstr: jobstr })
|
|
194
|
-
|
|
195
|
-
@capsule.redis do |conn|
|
|
196
|
-
conn.call('ZADD', Keys::DEAD, now.to_s, jobstr)
|
|
197
|
-
end
|
|
197
|
+
DeadSet.new.kill_raw(jobstr)
|
|
198
198
|
uow.acknowledge
|
|
199
199
|
nil
|
|
200
200
|
end
|
data/lib/wurk/profile_set.rb
CHANGED
|
@@ -21,16 +21,23 @@ module Wurk
|
|
|
21
21
|
|
|
22
22
|
def size = @keys.size
|
|
23
23
|
|
|
24
|
+
# HMGET of the metadata fields only, pipelined into one round-trip:
|
|
25
|
+
# HGETALL would also pull each profile's `data` field — the multi-MB
|
|
26
|
+
# gzipped blob — through Redis for every list render.
|
|
27
|
+
METADATA_FIELDS = %w[jid type token size elapsed started_at].freeze
|
|
28
|
+
|
|
24
29
|
def each
|
|
25
30
|
return enum_for(:each) unless block_given?
|
|
26
31
|
|
|
27
|
-
Wurk.redis do |conn|
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
hash = raw.is_a?(Hash) ? raw : raw.each_slice(2).to_h
|
|
31
|
-
yield ProfileRecord.new(hash) unless hash.empty?
|
|
32
|
+
rows = Wurk.redis do |conn|
|
|
33
|
+
conn.pipelined do |pipe|
|
|
34
|
+
@keys.each { |key| pipe.call('HMGET', key, *METADATA_FIELDS) }
|
|
32
35
|
end
|
|
33
36
|
end
|
|
37
|
+
rows.each do |values|
|
|
38
|
+
hash = METADATA_FIELDS.zip(Array(values)).to_h
|
|
39
|
+
yield ProfileRecord.new(hash) unless hash['jid'].nil?
|
|
40
|
+
end
|
|
34
41
|
end
|
|
35
42
|
end
|
|
36
43
|
|
|
@@ -39,6 +46,14 @@ module Wurk
|
|
|
39
46
|
class ProfileRecord
|
|
40
47
|
attr_reader :jid, :type, :token, :size, :elapsed
|
|
41
48
|
|
|
49
|
+
# Fetch the stored gzipped blob for a profile storage key ("<token>-<jid>")
|
|
50
|
+
# straight from Redis, without materializing the whole record — the Profiles
|
|
51
|
+
# data endpoint streams it to the browser as-is. nil if the HASH is gone.
|
|
52
|
+
# Owns the `data` HASH-field name so web callers don't hardcode the schema.
|
|
53
|
+
def self.data_for(key)
|
|
54
|
+
Wurk.redis { |conn| conn.call('HGET', key, 'data') }
|
|
55
|
+
end
|
|
56
|
+
|
|
42
57
|
def initialize(hash)
|
|
43
58
|
@hash = hash
|
|
44
59
|
@jid = hash['jid']
|
|
@@ -59,7 +74,7 @@ module Wurk
|
|
|
59
74
|
# bytes — the web layer streams them straight to the browser with a gzip
|
|
60
75
|
# Content-Encoding. Returns nil if the HASH expired between list and read.
|
|
61
76
|
def data
|
|
62
|
-
|
|
77
|
+
self.class.data_for(key)
|
|
63
78
|
end
|
|
64
79
|
end
|
|
65
80
|
end
|
data/lib/wurk/profiler.rb
CHANGED
|
@@ -50,11 +50,13 @@ module Wurk
|
|
|
50
50
|
key = profile_key(token, jid)
|
|
51
51
|
gz = gzip(gecko_json)
|
|
52
52
|
with_pool(pool) do |conn|
|
|
53
|
-
conn.
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
53
|
+
conn.pipelined do |pipe|
|
|
54
|
+
pipe.call('HSET', key, 'jid', jid, 'type', type, 'token', token,
|
|
55
|
+
'started_at', started_at.to_i, 'elapsed', elapsed_ms.to_i,
|
|
56
|
+
'size', gz.bytesize, 'sid', sid.to_s, 'data', gz)
|
|
57
|
+
pipe.call('EXPIRE', key, TTL)
|
|
58
|
+
pipe.call('ZADD', Keys::PROFILES, (now + TTL).to_i, key)
|
|
59
|
+
end
|
|
58
60
|
end
|
|
59
61
|
key
|
|
60
62
|
end
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Wurk
|
|
4
|
+
# Coordinates how Wurk boots inside a Rails host: whether this process should
|
|
5
|
+
# boot at all, whether it may fork the swarm, and the boot itself. The Railtie
|
|
6
|
+
# owns only the Rails hooks (config namespace + server_mode initializer +
|
|
7
|
+
# after_initialize) and delegates every decision and side effect here, so the
|
|
8
|
+
# boot policy stays pure and unit-testable without the Railtie DSL.
|
|
9
|
+
# See docs/idea/03-process-model.md for the exact ordering.
|
|
10
|
+
module RailsBoot
|
|
11
|
+
module_function
|
|
12
|
+
|
|
13
|
+
# Invoked from the `wurk.server_mode` initializer, before config/initializers
|
|
14
|
+
# load. This Rails process forks the workers, so it IS the server. Enter
|
|
15
|
+
# server mode now — otherwise the app's `Sidekiq.configure_server` blocks
|
|
16
|
+
# gate on `config.server?` (still false) and are silently dropped. A process
|
|
17
|
+
# that won't run workers (skip_boot?, or one that refuses to boot under a
|
|
18
|
+
# preforking web server) is not a server.
|
|
19
|
+
def enter_server_mode_if_serving(app = ::Rails.application)
|
|
20
|
+
return if skip_boot?
|
|
21
|
+
return if boot_action(app) == :refuse
|
|
22
|
+
|
|
23
|
+
Wurk.enter_server_mode
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
# Invoked from after_initialize, once the host app has fully initialized.
|
|
27
|
+
def boot(app = ::Rails.application)
|
|
28
|
+
return if skip_boot?
|
|
29
|
+
|
|
30
|
+
case boot_action(app)
|
|
31
|
+
when :fork then boot_swarm
|
|
32
|
+
when :embed then boot_embedded
|
|
33
|
+
when :refuse then refuse_preforking_boot
|
|
34
|
+
end
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
# A process that won't run workers isn't a server: skip both server mode
|
|
38
|
+
# and the swarm boot. Console mode is detected reliably here — the console
|
|
39
|
+
# command file defines ::Rails::Console before initializers run.
|
|
40
|
+
def skip_boot?
|
|
41
|
+
ENV['WURK_DISABLED'] == '1' ||
|
|
42
|
+
building? ||
|
|
43
|
+
defined?(::Rails::Console) ||
|
|
44
|
+
::Rails.env.test?
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
# What boot should do once skip_boot? is false. Pure — reads env / loaded
|
|
48
|
+
# constants / host config, no side effects — so the boot decision is
|
|
49
|
+
# unit-testable without forking:
|
|
50
|
+
# :fork — safe to fork the swarm in the background (the default).
|
|
51
|
+
# :refuse — a preforking web server owns process forking here; don't.
|
|
52
|
+
# :embed — host opted into in-process threads-only via embed_in_web.
|
|
53
|
+
def boot_action(app = ::Rails.application)
|
|
54
|
+
return :fork unless preforking_web_server?
|
|
55
|
+
|
|
56
|
+
embed_in_web?(app) ? :embed : :refuse
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
# Preforking / clustered web servers (Puma cluster, Unicorn, Passenger)
|
|
60
|
+
# fork their own worker processes. Forking the swarm from `after_initialize`
|
|
61
|
+
# in one of them is the highest-risk boot path: without app preloading every
|
|
62
|
+
# server-worker re-runs the hook and forks its own full swarm (N×
|
|
63
|
+
# oversubscription); with preloading the swarm supervisor ends up entangled
|
|
64
|
+
# with the server's own fork/signal supervision. Detect the common three.
|
|
65
|
+
def preforking_web_server?
|
|
66
|
+
return true if defined?(::PhusionPassenger)
|
|
67
|
+
return true if defined?(::Unicorn)
|
|
68
|
+
|
|
69
|
+
puma_cluster?
|
|
70
|
+
end
|
|
71
|
+
|
|
72
|
+
# Puma only preforks in cluster mode (workers > 0); single mode is threaded
|
|
73
|
+
# and safe to co-host the swarm. Best-effort: a missed cluster falls through
|
|
74
|
+
# to the historical fork path; an over-eager match is escapable via
|
|
75
|
+
# embed_in_web / WURK_DISABLED / running the swarm as its own process.
|
|
76
|
+
def puma_cluster?
|
|
77
|
+
return false unless defined?(::Puma)
|
|
78
|
+
|
|
79
|
+
count = puma_worker_count
|
|
80
|
+
count.is_a?(Integer) && count.positive?
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
# Worker count from Puma's parsed CLI config when the server booted it, else
|
|
84
|
+
# from WEB_CONCURRENCY (the near-universal convention for Puma workers).
|
|
85
|
+
# Guarded — the accessor and options shape vary across Puma versions.
|
|
86
|
+
def puma_worker_count
|
|
87
|
+
cfg = ::Puma.cli_config if ::Puma.respond_to?(:cli_config)
|
|
88
|
+
workers = cfg&.options&.[](:workers)
|
|
89
|
+
return workers.to_i if workers
|
|
90
|
+
|
|
91
|
+
ENV['WEB_CONCURRENCY']&.to_i
|
|
92
|
+
rescue StandardError
|
|
93
|
+
nil
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
# Host opt-in (config/application.rb): run workers as in-process threads
|
|
97
|
+
# instead of forking, like Sidekiq embedded. Set it in application.rb, not
|
|
98
|
+
# an initializer — server mode is decided before initializers load.
|
|
99
|
+
def embed_in_web?(app = ::Rails.application)
|
|
100
|
+
app.config.wurk&.embed_in_web == true
|
|
101
|
+
rescue StandardError
|
|
102
|
+
false
|
|
103
|
+
end
|
|
104
|
+
|
|
105
|
+
def boot_swarm
|
|
106
|
+
swarm = Wurk::Swarm.new(topology: Wurk.configuration.topology,
|
|
107
|
+
shutdown_timeout: Wurk.configuration[:timeout] || Swarm::DEFAULT_SHUTDOWN_TIMEOUT)
|
|
108
|
+
# Co-hosted in the web process (e.g. Puma single mode): the host owns the
|
|
109
|
+
# process-wide TERM/INT traps. Installing the swarm's own would hijack
|
|
110
|
+
# them — a deploy TERM would drain the swarm but never stop the HTTP
|
|
111
|
+
# server. Let the host keep signal ownership and drain the swarm on its
|
|
112
|
+
# graceful exit (same contract as boot_embedded).
|
|
113
|
+
swarm.boot(install_signals: false)
|
|
114
|
+
at_exit { swarm.shutdown }
|
|
115
|
+
# supervise must still run somewhere or crashed children never respawn and
|
|
116
|
+
# memory checks never fire. A background thread keeps the host's main
|
|
117
|
+
# thread free to serve HTTP.
|
|
118
|
+
Thread.new do
|
|
119
|
+
swarm.supervise
|
|
120
|
+
rescue StandardError => e
|
|
121
|
+
logger.error { "wurk supervisor thread died: #{e.class}: #{e.message}" }
|
|
122
|
+
end
|
|
123
|
+
end
|
|
124
|
+
|
|
125
|
+
# Sidekiq-embedded parity: a threads-only worker inside the web process, no
|
|
126
|
+
# fork. Redis validation failure keeps the host serving HTTP (log + carry
|
|
127
|
+
# on). at_exit drains on the graceful shutdown the web server runs on TERM.
|
|
128
|
+
def boot_embedded
|
|
129
|
+
instance = Wurk::Embedded.new(Wurk.configuration)
|
|
130
|
+
instance.run
|
|
131
|
+
at_exit { instance.stop }
|
|
132
|
+
logger.info { 'wurk: running embedded in the web process (config.wurk.embed_in_web) — threads only, no fork' }
|
|
133
|
+
instance
|
|
134
|
+
rescue StandardError => e
|
|
135
|
+
logger.error { "wurk: embedded boot failed: #{e.class}: #{e.message}" }
|
|
136
|
+
nil
|
|
137
|
+
end
|
|
138
|
+
|
|
139
|
+
def refuse_preforking_boot
|
|
140
|
+
logger.warn { <<~MSG }
|
|
141
|
+
wurk: preforking web server detected (Puma cluster / Unicorn / Passenger).
|
|
142
|
+
Refusing to fork the worker swarm from a process that forks its own workers.
|
|
143
|
+
Run the swarm as its own process instead:
|
|
144
|
+
|
|
145
|
+
bundle exec wurkswarm # forked swarm, real parallelism
|
|
146
|
+
bundle exec wurk # single process, thread pool
|
|
147
|
+
|
|
148
|
+
Or run workers inside this web process (threads only, no fork):
|
|
149
|
+
|
|
150
|
+
# config/application.rb
|
|
151
|
+
config.wurk.embed_in_web = true
|
|
152
|
+
|
|
153
|
+
Already running workers elsewhere? Set WURK_DISABLED=1 here to silence this.
|
|
154
|
+
MSG
|
|
155
|
+
end
|
|
156
|
+
|
|
157
|
+
def logger
|
|
158
|
+
Wurk.configuration.logger
|
|
159
|
+
end
|
|
160
|
+
|
|
161
|
+
# A build/precompile step must never fork the swarm (#247). The default
|
|
162
|
+
# Rails Dockerfile runs `SECRET_KEY_BASE_DUMMY=1 ./bin/rails
|
|
163
|
+
# assets:precompile`; that loads `:environment` → fires after_initialize,
|
|
164
|
+
# but there's no Redis during `docker build`, so a fork would hang/fail the
|
|
165
|
+
# build. Same for other env-loading rake tasks (db:prepare, db:migrate).
|
|
166
|
+
# The real server path is unaffected: `rails server` / `puma` boot through
|
|
167
|
+
# Rails::Command, not Rake, and don't set the dummy secret.
|
|
168
|
+
def building?
|
|
169
|
+
return true if ENV.key?('SECRET_KEY_BASE_DUMMY')
|
|
170
|
+
|
|
171
|
+
defined?(::Rake) && ::Rake.application.top_level_tasks.any?
|
|
172
|
+
rescue StandardError
|
|
173
|
+
false
|
|
174
|
+
end
|
|
175
|
+
end
|
|
176
|
+
end
|