wurk 1.1.0 → 1.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +25 -0
- data/app/controllers/concerns/wurk/same_origin_guard.rb +40 -0
- data/app/controllers/concerns/wurk/sse_streaming.rb +48 -0
- data/app/controllers/concerns/wurk/stream_concurrency_guard.rb +53 -0
- data/app/controllers/wurk/api/pagination.rb +60 -11
- data/app/controllers/wurk/api/serializers.rb +5 -1
- data/app/controllers/wurk/api_controller.rb +51 -70
- data/app/controllers/wurk/application_controller.rb +26 -0
- data/app/controllers/wurk/dashboard_controller.rb +23 -1
- data/app/controllers/wurk/extensions_controller.rb +3 -12
- data/app/controllers/wurk/profiles_controller.rb +6 -1
- data/lib/wurk/batch/death_handler.rb +13 -0
- data/lib/wurk/batch/server_middleware.rb +9 -5
- data/lib/wurk/batch.rb +3 -0
- data/lib/wurk/capsule.rb +34 -21
- data/lib/wurk/cli.rb +16 -3
- data/lib/wurk/client/buffered.rb +30 -10
- data/lib/wurk/client.rb +45 -1
- data/lib/wurk/component.rb +23 -9
- data/lib/wurk/configuration.rb +83 -18
- data/lib/wurk/cron.rb +13 -1
- data/lib/wurk/dead_set.rb +16 -1
- data/lib/wurk/engine.rb +17 -1
- data/lib/wurk/fetcher/reliable.rb +46 -13
- data/lib/wurk/fetcher.rb +5 -0
- data/lib/wurk/health.rb +74 -22
- data/lib/wurk/history.rb +4 -21
- data/lib/wurk/job_retry.rb +3 -5
- data/lib/wurk/launcher.rb +26 -3
- data/lib/wurk/limiter/server_middleware.rb +16 -1
- data/lib/wurk/limiter.rb +6 -5
- data/lib/wurk/lua/loader.rb +11 -5
- data/lib/wurk/lua.rb +100 -9
- data/lib/wurk/manager.rb +59 -19
- data/lib/wurk/metrics/history.rb +24 -38
- data/lib/wurk/metrics/queue_rollup.rb +4 -21
- data/lib/wurk/metrics/rollup.rb +4 -21
- data/lib/wurk/processor.rb +8 -8
- data/lib/wurk/profile_set.rb +21 -6
- data/lib/wurk/profiler.rb +7 -5
- data/lib/wurk/rails_boot.rb +176 -0
- data/lib/wurk/railtie.rb +19 -47
- data/lib/wurk/redis_connection.rb +6 -9
- data/lib/wurk/redis_pool.rb +148 -28
- data/lib/wurk/scheduled.rb +35 -12
- data/lib/wurk/swarm/backoff.rb +70 -0
- data/lib/wurk/swarm/child_boot.rb +92 -13
- data/lib/wurk/swarm/orphan_guard.rb +105 -0
- data/lib/wurk/swarm/restart.rb +196 -0
- data/lib/wurk/swarm.rb +194 -78
- data/lib/wurk/timer_loop.rb +49 -0
- data/lib/wurk/version.rb +1 -1
- data/lib/wurk/web/extension.rb +4 -1
- data/lib/wurk/web/pool_scope.rb +46 -0
- data/lib/wurk/web/rack_app.rb +2 -1
- data/lib/wurk/web/search.rb +77 -18
- data/lib/wurk/web.rb +1 -0
- data/lib/wurk/worker/setter.rb +6 -1
- data/lib/wurk.rb +6 -2
- data/vendor/assets/dashboard/assets/ArgsValue-DYfBiXrJ.js +1 -0
- data/vendor/assets/dashboard/assets/BatchDetail-DTZ2HzcD.js +1 -0
- data/vendor/assets/dashboard/assets/Batches-BrnXA332.js +1 -0
- data/vendor/assets/dashboard/assets/Busy-C57G8Xb3.js +1 -0
- data/vendor/assets/dashboard/assets/Cron-DlyH88oo.js +1 -0
- data/vendor/assets/dashboard/assets/Dashboard-B887pxWf.js +1 -0
- data/vendor/assets/dashboard/assets/Dead-BPA7gs-X.js +1 -0
- data/vendor/assets/dashboard/assets/Extension-Bunf6XuU.js +1 -0
- data/vendor/assets/dashboard/assets/FilterBox-FCDi4ZCU.js +1 -0
- data/vendor/assets/dashboard/assets/JobDetailModal-DuMdKUMm.js +2 -0
- data/vendor/assets/dashboard/assets/Limiters-Br0aCPMK.js +1 -0
- data/vendor/assets/dashboard/assets/Metrics-DxBmuywH.js +1 -0
- data/vendor/assets/dashboard/assets/Modal-t4FI_LaY.js +1 -0
- data/vendor/assets/dashboard/assets/PageHeader-CsDvJSOA.js +1 -0
- data/vendor/assets/dashboard/assets/Profiles-Bkhoqjlq.js +1 -0
- data/vendor/assets/dashboard/assets/Queues-BvhA-vfI.js +1 -0
- data/vendor/assets/dashboard/assets/Retries-JEpB-1Yl.js +1 -0
- data/vendor/assets/dashboard/assets/Scheduled-DN_FbSwP.js +1 -0
- data/vendor/assets/dashboard/assets/Search-DJuK0YCJ.js +1 -0
- data/vendor/assets/dashboard/assets/Skeleton-DOYDkzg1.js +1 -0
- data/vendor/assets/dashboard/assets/charts-CVK0zAnC.js +1 -0
- data/vendor/assets/dashboard/assets/index-BnPX9Ptn.css +1 -0
- data/vendor/assets/dashboard/assets/index-CZTcs-pM.js +141 -0
- data/vendor/assets/dashboard/assets/useResetPageOnEmpty-CoZU4b3a.js +1 -0
- data/vendor/assets/dashboard/assets/useSort-wQcnbdsa.js +1 -0
- data/vendor/assets/dashboard/index.html +3 -3
- data/vendor/assets/dashboard/wurk-manifest.json +2 -2
- metadata +55 -26
- data/vendor/assets/dashboard/assets/ArgsValue-BUqJa-eG.js +0 -1
- data/vendor/assets/dashboard/assets/BatchDetail-C5dkqAzp.js +0 -1
- data/vendor/assets/dashboard/assets/Batches-bgkXn8tc.js +0 -1
- data/vendor/assets/dashboard/assets/Busy-QSHBFXhy.js +0 -1
- data/vendor/assets/dashboard/assets/Cron-CUHppvTA.js +0 -1
- data/vendor/assets/dashboard/assets/Dashboard-CzPudckV.js +0 -1
- data/vendor/assets/dashboard/assets/Dead-BIq4Nz_i.js +0 -1
- data/vendor/assets/dashboard/assets/Extension-CW36i9R1.js +0 -1
- data/vendor/assets/dashboard/assets/JobDetailModal-BhNdWSp7.js +0 -2
- data/vendor/assets/dashboard/assets/Limiters-CiI_DfUK.js +0 -1
- data/vendor/assets/dashboard/assets/Metrics-BzZ8ugms.js +0 -1
- data/vendor/assets/dashboard/assets/Modal-DzVfgsSF.js +0 -1
- data/vendor/assets/dashboard/assets/PageHeader-Dp3qhX3e.js +0 -1
- data/vendor/assets/dashboard/assets/Profiles-BTtIlTdR.js +0 -1
- data/vendor/assets/dashboard/assets/Queues-BuXoxQ4W.js +0 -1
- data/vendor/assets/dashboard/assets/Retries-D4HAPaOQ.js +0 -1
- data/vendor/assets/dashboard/assets/Scheduled-6NCZYVJh.js +0 -1
- data/vendor/assets/dashboard/assets/Search-JaB_-52c.js +0 -1
- data/vendor/assets/dashboard/assets/charts-6uvCyY0x.js +0 -1
- data/vendor/assets/dashboard/assets/i18n-gIeA5VLo.js +0 -1
- data/vendor/assets/dashboard/assets/index-BDG9tvBA.css +0 -1
- data/vendor/assets/dashboard/assets/index-DWfWAGBc.js +0 -141
- data/vendor/assets/dashboard/assets/useJobSetActions-DTDaAoWr.js +0 -1
- data/vendor/assets/dashboard/assets/usePageParam-BcAvRko-.js +0 -1
- data/vendor/assets/dashboard/assets/useSort-D5Am4bGq.js +0 -1
data/lib/wurk/railtie.rb
CHANGED
|
@@ -1,60 +1,32 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
require 'rails/railtie'
|
|
4
|
+
require 'active_support/ordered_options'
|
|
5
|
+
require_relative 'rails_boot'
|
|
4
6
|
|
|
5
7
|
module Wurk
|
|
6
|
-
#
|
|
7
|
-
#
|
|
8
|
+
# Rails integration surface only: register the host-facing `config.wurk`
|
|
9
|
+
# namespace and wire the two boot hooks to Wurk::RailsBoot. All boot policy
|
|
10
|
+
# (skip/fork/embed/refuse, prefork detection) and execution lives there — this
|
|
11
|
+
# class exists solely to invoke the coordinator from the Rails lifecycle.
|
|
8
12
|
# See docs/idea/03-process-model.md for the exact ordering.
|
|
9
13
|
class Railtie < ::Rails::Railtie
|
|
10
|
-
#
|
|
11
|
-
#
|
|
12
|
-
#
|
|
13
|
-
#
|
|
14
|
-
#
|
|
15
|
-
|
|
16
|
-
Wurk.enter_server_mode unless Wurk::Railtie.skip_boot?
|
|
17
|
-
end
|
|
14
|
+
# Host-facing config namespace. Pre-created so a host can write
|
|
15
|
+
# `config.wurk.embed_in_web = true` in config/application.rb without a
|
|
16
|
+
# NoMethodError — the setter needs the OrderedOptions to already exist.
|
|
17
|
+
# Shared with the application config via Rails' Railtie::Configuration
|
|
18
|
+
# @@options, so `Rails.application.config.wurk` reads back the same object.
|
|
19
|
+
config.wurk = ::ActiveSupport::OrderedOptions.new
|
|
18
20
|
|
|
19
|
-
config
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
# Embedded mode keeps the Rails process serving HTTP; supervise must
|
|
25
|
-
# run somewhere or signal_queue never drains, crashed children never
|
|
26
|
-
# respawn, and memory pressure checks never fire. Background thread
|
|
27
|
-
# leaves the host's main thread free for Rails.
|
|
28
|
-
Thread.new do
|
|
29
|
-
swarm.supervise
|
|
30
|
-
rescue StandardError => e
|
|
31
|
-
Wurk.configuration.logger.error { "wurk supervisor thread died: #{e.class}: #{e.message}" }
|
|
32
|
-
end
|
|
33
|
-
end
|
|
34
|
-
|
|
35
|
-
# A process that won't run workers isn't a server: skip both server mode
|
|
36
|
-
# and the swarm boot. Console mode is detected reliably here — the console
|
|
37
|
-
# command file defines ::Rails::Console before initializers run.
|
|
38
|
-
def self.skip_boot?
|
|
39
|
-
ENV['WURK_DISABLED'] == '1' ||
|
|
40
|
-
building? ||
|
|
41
|
-
defined?(::Rails::Console) ||
|
|
42
|
-
::Rails.env.test?
|
|
21
|
+
# Enter server mode BEFORE config/initializers load — otherwise the app's
|
|
22
|
+
# `Sidekiq.configure_server` blocks gate on `config.server?` (still false)
|
|
23
|
+
# and are silently dropped. RailsBoot decides whether this process serves.
|
|
24
|
+
initializer 'wurk.server_mode', before: :load_config_initializers do |app|
|
|
25
|
+
Wurk::RailsBoot.enter_server_mode_if_serving(app)
|
|
43
26
|
end
|
|
44
27
|
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
# assets:precompile`; that loads `:environment` → fires after_initialize,
|
|
48
|
-
# but there's no Redis during `docker build`, so a fork would hang/fail the
|
|
49
|
-
# build. Same for other env-loading rake tasks (db:prepare, db:migrate).
|
|
50
|
-
# The real server path is unaffected: `rails server` / `puma` boot through
|
|
51
|
-
# Rails::Command, not Rake, and don't set the dummy secret.
|
|
52
|
-
def self.building?
|
|
53
|
-
return true if ENV.key?('SECRET_KEY_BASE_DUMMY')
|
|
54
|
-
|
|
55
|
-
defined?(::Rake) && ::Rake.application.top_level_tasks.any?
|
|
56
|
-
rescue StandardError
|
|
57
|
-
false
|
|
28
|
+
config.after_initialize do |app|
|
|
29
|
+
Wurk::RailsBoot.boot(app)
|
|
58
30
|
end
|
|
59
31
|
end
|
|
60
32
|
end
|
|
@@ -10,21 +10,18 @@ module Wurk
|
|
|
10
10
|
# (connection_pool-backed, redis-client adapter), which is `.with`-compatible
|
|
11
11
|
# with everything that expects a Sidekiq pool.
|
|
12
12
|
#
|
|
13
|
-
# Accepts Sidekiq's option keys (`url`, `size`, `pool_timeout`, `name`)
|
|
14
|
-
#
|
|
15
|
-
#
|
|
13
|
+
# Accepts Sidekiq's option keys (`url`, `size`, `pool_timeout`, `name`) plus
|
|
14
|
+
# the socket knobs (`connect_timeout`/`read_timeout`/`write_timeout`/
|
|
15
|
+
# `reconnect_attempts`/`driver`), string- or symbol-keyed. Anything omitted
|
|
16
|
+
# falls back to RedisPool's defaults (URL = ENV["REDIS_URL"] or
|
|
17
|
+
# redis://localhost:6379/0).
|
|
16
18
|
module RedisConnection
|
|
17
19
|
# Housekeeping/standalone default; per-capsule pools size to concurrency.
|
|
18
20
|
DEFAULT_POOL_SIZE = 10
|
|
19
21
|
|
|
20
22
|
def self.create(options = {})
|
|
21
23
|
opts = options.transform_keys(&:to_sym)
|
|
22
|
-
RedisPool.new(
|
|
23
|
-
size: opts[:size] || DEFAULT_POOL_SIZE,
|
|
24
|
-
url: opts[:url] || RedisPool::DEFAULT_URL,
|
|
25
|
-
timeout: opts[:pool_timeout] || RedisPool::DEFAULT_TIMEOUT,
|
|
26
|
-
name: opts[:name] || RedisPool::DEFAULT_NAME
|
|
27
|
-
)
|
|
24
|
+
RedisPool.new(size: opts.delete(:size) || DEFAULT_POOL_SIZE, **opts)
|
|
28
25
|
end
|
|
29
26
|
end
|
|
30
27
|
end
|
data/lib/wurk/redis_pool.rb
CHANGED
|
@@ -9,41 +9,86 @@ module Wurk
|
|
|
9
9
|
# across forks: the parent closes the pool before fork, each child opens a
|
|
10
10
|
# fresh one (see docs/idea/03-process-model.md, steps 3 and 5).
|
|
11
11
|
#
|
|
12
|
-
#
|
|
13
|
-
#
|
|
12
|
+
# #with absorbs transient Redis failures (production incident #101):
|
|
13
|
+
# * READONLY / NOREPLICAS / UNBLOCKED — a failover happened; close and retry
|
|
14
|
+
# once immediately so redis-client redials the new primary (spec §26).
|
|
15
|
+
# * ConnectionError (incl. CannotConnect / Read- / WriteTimeout) — a blip;
|
|
16
|
+
# close and retry with exponential backoff up to CONN_MAX_ATTEMPTS, then
|
|
17
|
+
# raise. At-least-once tolerates the rare duplicate and JobRetry re-runs
|
|
18
|
+
# the job anyway, so retrying at the pool layer is strictly better.
|
|
19
|
+
# * ConnectionPool::TimeoutError — checkout starved; retry once after a
|
|
20
|
+
# short jittered pause, then raise (sizing is the fix, not queuing).
|
|
21
|
+
# Every retry and final give-up is reported through the injected `on_error`
|
|
22
|
+
# telemetry hook (Wurk::Configuration#on_redis_error).
|
|
14
23
|
class RedisPool
|
|
15
|
-
DEFAULT_URL
|
|
16
|
-
|
|
17
|
-
DEFAULT_NAME = 'default'
|
|
24
|
+
DEFAULT_URL = ENV.fetch('REDIS_URL', 'redis://localhost:6379/0')
|
|
25
|
+
DEFAULT_NAME = 'default'
|
|
18
26
|
|
|
19
|
-
#
|
|
20
|
-
|
|
21
|
-
|
|
27
|
+
# ConnectionPool checkout wait — how long #with blocks for a free slot.
|
|
28
|
+
DEFAULT_POOL_TIMEOUT = 1.0
|
|
29
|
+
|
|
30
|
+
# Socket-level timeouts handed to RedisClient, split apart from the checkout
|
|
31
|
+
# wait above. read/write are deliberately wider than connect so a briefly-
|
|
32
|
+
# slow-but-alive Redis (RDB fork pause, a large BLMOVE payload) doesn't
|
|
33
|
+
# spuriously ReadTimeout — the production incident (#101) the single
|
|
34
|
+
# dual-use timeout caused. reconnect_attempts re-dials a dropped socket once.
|
|
35
|
+
DEFAULT_CONNECT_TIMEOUT = 1.0
|
|
36
|
+
DEFAULT_READ_TIMEOUT = 2.5
|
|
37
|
+
DEFAULT_WRITE_TIMEOUT = 2.5
|
|
38
|
+
DEFAULT_RECONNECT_ATTEMPTS = 1
|
|
39
|
+
|
|
40
|
+
# Server-side messages where the connection is closed and the block retried
|
|
41
|
+
# exactly once. READONLY is itself a RedisClient::ConnectionError subclass,
|
|
42
|
+
# so this message match must be tested BEFORE the generic ConnectionError
|
|
43
|
+
# backoff below (otherwise a failover would sleep instead of redialing).
|
|
22
44
|
RETRYABLE_MSG = /\A(READONLY|NOREPLICAS|UNBLOCKED)/
|
|
23
45
|
|
|
24
|
-
|
|
46
|
+
# ConnectionError backoff: CONN_MAX_ATTEMPTS total tries, sleeping
|
|
47
|
+
# (BASE * 2**attempt) + rand*JITTER before each retry. The 1.0s + 2.0s pair
|
|
48
|
+
# rides out a sub-4s blip; the jitter de-syncs a fleet reconnecting at once.
|
|
49
|
+
CONN_MAX_ATTEMPTS = 3
|
|
50
|
+
CONN_BACKOFF_BASE = 0.5
|
|
51
|
+
CONN_BACKOFF_JITTER = 0.25
|
|
52
|
+
|
|
53
|
+
# Checkout-timeout retry: one retry after a jittered pause in
|
|
54
|
+
# [POOL_RETRY_MIN, POOL_RETRY_MIN + POOL_RETRY_SPREAD). No loop — sustained
|
|
55
|
+
# checkout starvation is a sizing bug, not something to queue behind.
|
|
56
|
+
POOL_RETRY_MIN = 0.1
|
|
57
|
+
POOL_RETRY_SPREAD = 0.2
|
|
58
|
+
|
|
59
|
+
attr_reader :size, :url, :name, :pool_timeout, :client_config
|
|
25
60
|
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
61
|
+
# Takes the standard Sidekiq `config.redis` hash: `pool_timeout` tunes the
|
|
62
|
+
# ConnectionPool checkout; `connect_timeout`/`read_timeout`/`write_timeout`/
|
|
63
|
+
# `reconnect_attempts` plus any other key (driver, ssl_params, …) forward
|
|
64
|
+
# verbatim to RedisClient.config. `on_error` is an optional callable fired
|
|
65
|
+
# per retry / final give-up with { error:, attempt:, retried:, pool: }.
|
|
66
|
+
def initialize(size:, name: DEFAULT_NAME, on_error: nil, **options)
|
|
67
|
+
@size = size
|
|
68
|
+
@name = name
|
|
69
|
+
@on_error = on_error
|
|
70
|
+
@pool_timeout = options.fetch(:pool_timeout, DEFAULT_POOL_TIMEOUT)
|
|
71
|
+
@client_config = build_client_config(options)
|
|
72
|
+
@url = @client_config[:url]
|
|
73
|
+
@pool = ConnectionPool.new(size: size, timeout: @pool_timeout) { build_client }
|
|
32
74
|
end
|
|
33
75
|
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
retry
|
|
76
|
+
# Checkout a connection and run the block. ConnectionPool::TimeoutError is
|
|
77
|
+
# raised by @pool.with *before* the block runs, so it is caught out here
|
|
78
|
+
# (the in-block #run rescue never sees it) — one retry, then raise.
|
|
79
|
+
def with(&block)
|
|
80
|
+
checkout_retried = false
|
|
81
|
+
begin
|
|
82
|
+
@pool.with { |conn| run(conn, &block) }
|
|
83
|
+
rescue ConnectionPool::TimeoutError => e
|
|
84
|
+
if checkout_retried
|
|
85
|
+
notify_error(e, attempt: 2, retried: false)
|
|
86
|
+
raise
|
|
46
87
|
end
|
|
88
|
+
checkout_retried = true
|
|
89
|
+
notify_error(e, attempt: 1, retried: true)
|
|
90
|
+
sleep(checkout_delay)
|
|
91
|
+
retry
|
|
47
92
|
end
|
|
48
93
|
end
|
|
49
94
|
|
|
@@ -51,17 +96,40 @@ module Wurk
|
|
|
51
96
|
@pool.shutdown { |conn| safe_close(conn) }
|
|
52
97
|
end
|
|
53
98
|
|
|
99
|
+
# Redis INFO parsed to a Hash, with this pool's own `size` / `available`
|
|
100
|
+
# slot counts merged in — one call gives a heartbeat both Redis health and
|
|
101
|
+
# local pool saturation. (Real Redis INFO has no `size`/`available` field.)
|
|
54
102
|
def info
|
|
55
103
|
with { |conn| parse_info(conn.call('INFO')) }
|
|
104
|
+
.merge('size' => @size, 'available' => available)
|
|
105
|
+
end
|
|
106
|
+
|
|
107
|
+
# Free (unchecked-out) slots right now. Local and cheap — no Redis round
|
|
108
|
+
# trip — so a monitor can poll it without perturbing the pool.
|
|
109
|
+
def available
|
|
110
|
+
@pool.available
|
|
56
111
|
end
|
|
57
112
|
|
|
58
113
|
private
|
|
59
114
|
|
|
115
|
+
# Socket config forwarded to RedisClient.config. Host-supplied keys win over
|
|
116
|
+
# the defaults; `pool_timeout` is dropped (it's a pool concern, not a socket
|
|
117
|
+
# one) and unknown keys pass straight through.
|
|
118
|
+
def build_client_config(options)
|
|
119
|
+
{
|
|
120
|
+
url: DEFAULT_URL,
|
|
121
|
+
connect_timeout: DEFAULT_CONNECT_TIMEOUT,
|
|
122
|
+
read_timeout: DEFAULT_READ_TIMEOUT,
|
|
123
|
+
write_timeout: DEFAULT_WRITE_TIMEOUT,
|
|
124
|
+
reconnect_attempts: DEFAULT_RECONNECT_ATTEMPTS
|
|
125
|
+
}.merge(options.except(:pool_timeout)).freeze
|
|
126
|
+
end
|
|
127
|
+
|
|
60
128
|
# Wrapped in the CompatClient decorator so `Sidekiq.redis { |c| c.smembers }`
|
|
61
129
|
# method-style commands work like Sidekiq 7+ (#204). Wurk's own code paths
|
|
62
130
|
# use #call, which the decorator forwards.
|
|
63
131
|
def build_client
|
|
64
|
-
RedisClientAdapter::CompatClient.new(RedisClient.config(
|
|
132
|
+
RedisClientAdapter::CompatClient.new(RedisClient.config(**@client_config).new_client)
|
|
65
133
|
end
|
|
66
134
|
|
|
67
135
|
def safe_close(conn)
|
|
@@ -70,6 +138,58 @@ module Wurk
|
|
|
70
138
|
nil
|
|
71
139
|
end
|
|
72
140
|
|
|
141
|
+
# Runs the block on the checked-out `conn`, retrying transient RedisClient
|
|
142
|
+
# errors in place: the same slot is reused across retries (redis-client
|
|
143
|
+
# redials a closed socket lazily), so a busy fetcher can't leak checkouts.
|
|
144
|
+
def run(conn)
|
|
145
|
+
attempts = 0
|
|
146
|
+
begin
|
|
147
|
+
attempts += 1
|
|
148
|
+
yield conn
|
|
149
|
+
rescue RedisClient::Error => e
|
|
150
|
+
plan = retry_plan(e, attempts)
|
|
151
|
+
raise if plan == :propagate
|
|
152
|
+
|
|
153
|
+
notify_error(e, attempt: attempts, retried: plan != :exhausted)
|
|
154
|
+
raise if plan == :exhausted
|
|
155
|
+
|
|
156
|
+
safe_close(conn)
|
|
157
|
+
sleep(backoff_delay(attempts)) if plan == :backoff
|
|
158
|
+
retry
|
|
159
|
+
end
|
|
160
|
+
end
|
|
161
|
+
|
|
162
|
+
# Pure classification of a RedisClient error against the attempt count:
|
|
163
|
+
# :failover → close + immediate retry (a primary swap)
|
|
164
|
+
# :backoff → close + sleep + retry (a connection blip)
|
|
165
|
+
# :exhausted → ConnectionError past the cap; report and give up
|
|
166
|
+
# :propagate → not transient (or a spent failover); raise as-is
|
|
167
|
+
def retry_plan(err, attempts)
|
|
168
|
+
if RETRYABLE_MSG.match?(err.message.to_s)
|
|
169
|
+
attempts > 1 ? :propagate : :failover
|
|
170
|
+
elsif err.is_a?(RedisClient::ConnectionError)
|
|
171
|
+
attempts >= CONN_MAX_ATTEMPTS ? :exhausted : :backoff
|
|
172
|
+
else
|
|
173
|
+
:propagate
|
|
174
|
+
end
|
|
175
|
+
end
|
|
176
|
+
|
|
177
|
+
def notify_error(error, attempt:, retried:)
|
|
178
|
+
return if @on_error.nil?
|
|
179
|
+
|
|
180
|
+
@on_error.call({ error:, attempt:, retried:, pool: @name })
|
|
181
|
+
rescue StandardError
|
|
182
|
+
nil
|
|
183
|
+
end
|
|
184
|
+
|
|
185
|
+
def backoff_delay(attempt)
|
|
186
|
+
(CONN_BACKOFF_BASE * (2**attempt)) + (rand * CONN_BACKOFF_JITTER)
|
|
187
|
+
end
|
|
188
|
+
|
|
189
|
+
def checkout_delay
|
|
190
|
+
POOL_RETRY_MIN + (rand * POOL_RETRY_SPREAD)
|
|
191
|
+
end
|
|
192
|
+
|
|
73
193
|
def parse_info(raw)
|
|
74
194
|
raw.to_s.each_line.with_object({}) do |line, h|
|
|
75
195
|
line = line.strip
|
data/lib/wurk/scheduled.rb
CHANGED
|
@@ -39,9 +39,7 @@ module Wurk
|
|
|
39
39
|
# client. `now` is captured once per set so a slow loop on one ZSET
|
|
40
40
|
# can't keep grabbing newly-scheduled jobs from a moving window.
|
|
41
41
|
def enqueue_jobs(sorted_sets = SETS)
|
|
42
|
-
|
|
43
|
-
sorted_sets.each { |sset| drain_set(conn, sset) }
|
|
44
|
-
end
|
|
42
|
+
sorted_sets.each { |sset| drain_set(sset) }
|
|
45
43
|
end
|
|
46
44
|
|
|
47
45
|
def terminate
|
|
@@ -50,18 +48,34 @@ module Wurk
|
|
|
50
48
|
|
|
51
49
|
private
|
|
52
50
|
|
|
53
|
-
|
|
51
|
+
# Pop under one checkout (the pooled EVALSHA loop), push outside it —
|
|
52
|
+
# `@client.push` checks out its own connection, so nesting it inside
|
|
53
|
+
# `@config.redis` would hold two checkouts from the same pool at once
|
|
54
|
+
# per job, halving effective pool concurrency under load.
|
|
55
|
+
def drain_set(sset)
|
|
54
56
|
now = real_time.to_s
|
|
55
57
|
loop do
|
|
56
58
|
break if @done
|
|
57
59
|
|
|
58
|
-
jobstr = Wurk::Lua::Loader.eval_cached(conn, :zpopbyscore, keys: [sset], argv: [now])
|
|
60
|
+
jobstr = @config.redis { |conn| Wurk::Lua::Loader.eval_cached(conn, :zpopbyscore, keys: [sset], argv: [now]) }
|
|
59
61
|
break unless jobstr
|
|
60
62
|
|
|
61
|
-
|
|
63
|
+
push_promoted(jobstr, sset)
|
|
62
64
|
end
|
|
63
65
|
end
|
|
64
66
|
|
|
67
|
+
# A raising `@client.push` (bad payload, transient Redis error) must not
|
|
68
|
+
# abort the drain and strand the remaining due jobs until the next poll —
|
|
69
|
+
# rescue per-job, report, continue. The already-popped job IS lost here
|
|
70
|
+
# (ZPOPBYSCORE removed it); that pop→push loss window is the default
|
|
71
|
+
# scheduler's known tradeoff — `reliable_scheduler!` (ReliableEnq) is the
|
|
72
|
+
# loss-free fix, so we don't re-engineer around it here.
|
|
73
|
+
def push_promoted(jobstr, sset)
|
|
74
|
+
@client.push(Wurk.load_json(jobstr))
|
|
75
|
+
rescue StandardError => e
|
|
76
|
+
handle_exception(e, { context: 'scheduler_promote', set: sset })
|
|
77
|
+
end
|
|
78
|
+
|
|
65
79
|
def real_time
|
|
66
80
|
::Process.clock_gettime(::Process::CLOCK_REALTIME)
|
|
67
81
|
end
|
|
@@ -78,6 +92,12 @@ module Wurk
|
|
|
78
92
|
class ReliableEnq
|
|
79
93
|
include Component
|
|
80
94
|
|
|
95
|
+
# Batched: each Lua call promotes at most this many members (the script
|
|
96
|
+
# runs atomically, so one giant sweep would stall Redis for every
|
|
97
|
+
# client). `promote` loops until a short batch signals the backlog is
|
|
98
|
+
# dry, stopping early on terminate.
|
|
99
|
+
PROMOTE_BATCH = 500
|
|
100
|
+
|
|
81
101
|
def initialize(container)
|
|
82
102
|
@config = container
|
|
83
103
|
@done = false
|
|
@@ -96,12 +116,15 @@ module Wurk
|
|
|
96
116
|
private
|
|
97
117
|
|
|
98
118
|
def promote(conn, sset)
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
119
|
+
loop do
|
|
120
|
+
promoted = Wurk::Lua::Loader.eval_cached(
|
|
121
|
+
conn,
|
|
122
|
+
:reliable_schedule_promote,
|
|
123
|
+
keys: [sset, Keys::QUEUES_SET],
|
|
124
|
+
argv: [real_time.to_s, Keys::QUEUE_PREFIX, real_ms.to_s, PROMOTE_BATCH.to_s]
|
|
125
|
+
).to_i
|
|
126
|
+
break if promoted < PROMOTE_BATCH || @done
|
|
127
|
+
end
|
|
105
128
|
end
|
|
106
129
|
|
|
107
130
|
def real_time
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Wurk
|
|
4
|
+
class Swarm
|
|
5
|
+
# Per-key exponential backoff timer with survival-based reset. Pure
|
|
6
|
+
# bookkeeping — no sleeping, no I/O — so the supervise thread never blocks
|
|
7
|
+
# on it: it records a failure, then asks `ready?` each tick and acts only
|
|
8
|
+
# once the delay has elapsed.
|
|
9
|
+
#
|
|
10
|
+
# Delay grows base → 2·base → 4·base … capped at `cap`. A key whose child
|
|
11
|
+
# survived at least `reset_after` seconds counts as a fresh first failure,
|
|
12
|
+
# so a slot that ran healthily for a while doesn't inherit an old crash
|
|
13
|
+
# streak. Keyed by slot index for crash-respawn and (in Swarm::Restart) for
|
|
14
|
+
# replacement-retry delays.
|
|
15
|
+
class Backoff
|
|
16
|
+
BASE = 1.0
|
|
17
|
+
CAP = 30.0
|
|
18
|
+
RESET_AFTER = 60.0
|
|
19
|
+
|
|
20
|
+
def initialize(base: BASE, cap: CAP, reset_after: RESET_AFTER, clock: nil)
|
|
21
|
+
@base = base
|
|
22
|
+
@cap = cap
|
|
23
|
+
@reset_after = reset_after
|
|
24
|
+
@clock = clock || -> { ::Process.clock_gettime(::Process::CLOCK_MONOTONIC) }
|
|
25
|
+
@streak = Hash.new(0)
|
|
26
|
+
@due_at = {}
|
|
27
|
+
end
|
|
28
|
+
|
|
29
|
+
# Record a failure and schedule the next attempt. `lifetime` is how long
|
|
30
|
+
# the failed child lived; >= reset_after resets the streak. Returns the
|
|
31
|
+
# delay applied.
|
|
32
|
+
def fail(key, lifetime: 0.0)
|
|
33
|
+
@streak[key] = 0 if lifetime >= @reset_after
|
|
34
|
+
@streak[key] += 1
|
|
35
|
+
delay = [@base * (2**(@streak[key] - 1)), @cap].min
|
|
36
|
+
@due_at[key] = now + delay
|
|
37
|
+
delay
|
|
38
|
+
end
|
|
39
|
+
|
|
40
|
+
# A retry is scheduled and not yet issued.
|
|
41
|
+
def pending?(key)
|
|
42
|
+
@due_at.key?(key)
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
# The scheduled delay has elapsed (or nothing is scheduled).
|
|
46
|
+
def ready?(key)
|
|
47
|
+
at = @due_at[key]
|
|
48
|
+
at.nil? || now >= at
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
# Mark the scheduled retry as issued without resetting the streak, so a
|
|
52
|
+
# child that crashes straight back escalates toward the cap.
|
|
53
|
+
def consume(key)
|
|
54
|
+
@due_at.delete(key)
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
# Forget the key entirely (slot retired or restart succeeded).
|
|
58
|
+
def clear(key)
|
|
59
|
+
@streak.delete(key)
|
|
60
|
+
@due_at.delete(key)
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
private
|
|
64
|
+
|
|
65
|
+
def now
|
|
66
|
+
@clock.call
|
|
67
|
+
end
|
|
68
|
+
end
|
|
69
|
+
end
|
|
70
|
+
end
|
|
@@ -3,6 +3,8 @@
|
|
|
3
3
|
require_relative '../component'
|
|
4
4
|
require_relative '../launcher'
|
|
5
5
|
require_relative '../fetcher/reliable'
|
|
6
|
+
require_relative '../lua'
|
|
7
|
+
require_relative 'orphan_guard'
|
|
6
8
|
|
|
7
9
|
module Wurk
|
|
8
10
|
class Swarm
|
|
@@ -21,11 +23,23 @@ module Wurk
|
|
|
21
23
|
|
|
22
24
|
CHILD_SIGNALS = { 'TERM' => :term, 'INT' => :term, 'TSTP' => :tstp, 'USR2' => :usr2 }.freeze
|
|
23
25
|
|
|
24
|
-
|
|
26
|
+
# `parent_pid` is captured by the swarm before it forks (race-free) and
|
|
27
|
+
# threaded through so OrphanGuard can tell "still supervised" from
|
|
28
|
+
# "reparented after the supervisor died". Defaults to the live parent for
|
|
29
|
+
# the non-swarm callers (tests) that construct a ChildBoot directly.
|
|
30
|
+
# `start_quiet:` — the swarm was TSTP-quieted before this child was
|
|
31
|
+
# forked (respawn/recycle during maintenance); boot the launcher already
|
|
32
|
+
# quieted so the replacement doesn't resume fetching. Delivered as a
|
|
33
|
+
# constructor flag, not a post-fork TSTP, because the signal would race
|
|
34
|
+
# the trap-reset window (default TSTP disposition suspends the child).
|
|
35
|
+
def initialize(config, slot, index, parent_pid: ::Process.ppid, start_quiet: false)
|
|
25
36
|
@config = config
|
|
26
37
|
@slot = slot
|
|
27
38
|
@index = index
|
|
28
|
-
@
|
|
39
|
+
@parent_pid = parent_pid
|
|
40
|
+
@start_quiet = start_quiet
|
|
41
|
+
@signal_read = nil
|
|
42
|
+
@signal_write = nil
|
|
29
43
|
end
|
|
30
44
|
|
|
31
45
|
def run
|
|
@@ -56,28 +70,71 @@ module Wurk
|
|
|
56
70
|
|
|
57
71
|
# Boot the launcher and block until shutdown. wait_loop joins the
|
|
58
72
|
# signal-dispatch thread, so the child can't fall through to `exit 0`
|
|
59
|
-
# mid-drain.
|
|
73
|
+
# mid-drain. Orphan protection is armed right AFTER launcher.run — the
|
|
74
|
+
# TERM handler is in place (so pdeathsig / the watchdog drain gracefully)
|
|
75
|
+
# and the managers are up (so a self-TERM can't race launcher.run). A
|
|
76
|
+
# parent that died during boot is still caught immediately: the watchdog's
|
|
77
|
+
# first getppid check sees the reparent and drains at once.
|
|
60
78
|
def run_launcher
|
|
61
79
|
launcher = Wurk::Launcher.new(@config)
|
|
62
80
|
install_signal_handlers(launcher)
|
|
63
81
|
launcher.run
|
|
82
|
+
launcher.quiet if @start_quiet
|
|
83
|
+
arm_orphan_guard
|
|
64
84
|
wait_loop(launcher)
|
|
65
85
|
end
|
|
66
86
|
|
|
67
|
-
|
|
68
|
-
|
|
87
|
+
def arm_orphan_guard
|
|
88
|
+
@orphan_guard = OrphanGuard.new(@parent_pid, logger: @config.logger)
|
|
89
|
+
@watchdog = @orphan_guard.arm
|
|
90
|
+
end
|
|
91
|
+
|
|
92
|
+
# Parent installed traps for TERM/INT/TSTP — the child needs its own
|
|
93
|
+
# behavior, not the parent's. USR2 too: the child owns log-reopen.
|
|
94
|
+
# USR1 (rolling restart) is a no-op in the child — trap it with a log
|
|
95
|
+
# so stray USR1s don't trigger default termination.
|
|
69
96
|
def reset_inherited_signals
|
|
70
|
-
%w[TERM INT TSTP
|
|
97
|
+
%w[TERM INT TSTP USR2].each { |s| ::Signal.trap(s, 'DEFAULT') }
|
|
98
|
+
# A genuine no-op in the child (rolling restart is the parent's job) —
|
|
99
|
+
# trap it empty so a stray USR1 can't fall through to default
|
|
100
|
+
# termination. Must NOT log: Logger synchronizes writes and would
|
|
101
|
+
# deadlock if the trap fired mid-write.
|
|
102
|
+
::Signal.trap('USR1') { nil }
|
|
103
|
+
rescue ArgumentError
|
|
104
|
+
nil
|
|
71
105
|
end
|
|
72
106
|
|
|
73
107
|
def reconnect_after_fork
|
|
74
108
|
@config.reset_redis_pools!
|
|
109
|
+
validate_redis!
|
|
110
|
+
reconnect_active_record
|
|
111
|
+
end
|
|
112
|
+
|
|
113
|
+
# Prove the child's fresh Redis socket reaches a live server before it
|
|
114
|
+
# starts fetching: one PING through RedisPool#with, which owns the
|
|
115
|
+
# retry+backoff (production incident #101), so a transient blip during
|
|
116
|
+
# boot rides out instead of racing straight into a dead pool. A PING that
|
|
117
|
+
# still fails past the wrapper's retries propagates and crashes the child
|
|
118
|
+
# (the swarm respawns it) rather than booting a worker that can't reach
|
|
119
|
+
# Redis. The same checkout eagerly primes every Lua script in one
|
|
120
|
+
# pipelined round-trip so the first EVALSHA hits a warm cache.
|
|
121
|
+
def validate_redis!
|
|
122
|
+
@config.redis_pool.with do |conn|
|
|
123
|
+
conn.call('PING')
|
|
124
|
+
Wurk::Lua::Loader.script_load_all(conn)
|
|
125
|
+
end
|
|
126
|
+
end
|
|
127
|
+
|
|
128
|
+
# AR reconnect is best-effort — a Redis-only worker with no database still
|
|
129
|
+
# runs — but the silent `rescue nil` here hid a real misconfiguration
|
|
130
|
+
# during the #101 audit, so warn loudly instead of swallowing.
|
|
131
|
+
def reconnect_active_record
|
|
75
132
|
return unless defined?(::ActiveRecord::Base)
|
|
76
133
|
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
134
|
+
::ActiveRecord::Base.establish_connection
|
|
135
|
+
rescue StandardError => e
|
|
136
|
+
@config.logger.warn do
|
|
137
|
+
"swarm child ##{@index} (#{::Process.pid}) ActiveRecord reconnect failed: #{e.class}: #{e.message}"
|
|
81
138
|
end
|
|
82
139
|
end
|
|
83
140
|
|
|
@@ -90,11 +147,29 @@ module Wurk
|
|
|
90
147
|
# below — for every entry point, not just the swarm.
|
|
91
148
|
end
|
|
92
149
|
|
|
150
|
+
# Self-pipe pattern (same as Wurk::CLI / Wurk::Swarm): the trap only
|
|
151
|
+
# writes the signal name to a pipe — never Thread::Queue#push, whose
|
|
152
|
+
# mutex a trap can deadlock against.
|
|
93
153
|
def install_signal_handlers(launcher)
|
|
94
|
-
|
|
154
|
+
@signal_read, @signal_write = ::IO.pipe
|
|
155
|
+
CHILD_SIGNALS.each_key do |sig|
|
|
156
|
+
::Signal.trap(sig) { emit_signal(sig) }
|
|
157
|
+
rescue ArgumentError
|
|
158
|
+
nil
|
|
159
|
+
end
|
|
95
160
|
@dispatcher = Thread.new { dispatch_signals(launcher) }
|
|
96
161
|
end
|
|
97
162
|
|
|
163
|
+
# Non-blocking self-pipe write from trap context: a blocking `puts` could
|
|
164
|
+
# stall the TERM drain if the pipe ever filled. `exception: false` returns
|
|
165
|
+
# :wait_writable instead of raising when full (the queued duplicate
|
|
166
|
+
# coalesces); a closed pipe during shutdown is ignored too.
|
|
167
|
+
def emit_signal(sig)
|
|
168
|
+
@signal_write.write_nonblock("#{sig}\n", exception: false)
|
|
169
|
+
rescue ::IOError, ::Errno::EPIPE, ::Errno::EBADF
|
|
170
|
+
nil
|
|
171
|
+
end
|
|
172
|
+
|
|
98
173
|
# TSTP/USR2 keep looping; TERM/INT run the full launcher.stop
|
|
99
174
|
# (which blocks on manager drain) and then return — wait_loop
|
|
100
175
|
# joins this thread, so the main child thread can't `exit 0`
|
|
@@ -102,10 +177,14 @@ module Wurk
|
|
|
102
177
|
# and the main thread would race past the unfinished managers.
|
|
103
178
|
def dispatch_signals(launcher)
|
|
104
179
|
loop do
|
|
105
|
-
|
|
180
|
+
@signal_read.wait_readable
|
|
181
|
+
sig = @signal_read.gets&.strip
|
|
182
|
+
break if sig.nil?
|
|
183
|
+
|
|
184
|
+
case CHILD_SIGNALS[sig]
|
|
106
185
|
when :term
|
|
107
186
|
launcher.stop
|
|
108
|
-
|
|
187
|
+
break
|
|
109
188
|
when :tstp then launcher.quiet
|
|
110
189
|
when :usr2 then reopen_logs
|
|
111
190
|
end
|