patient_http-sidekiq 1.2.0 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. checksums.yaml +4 -4
  2. data/ARCHITECTURE.md +31 -10
  3. data/CHANGELOG.md +32 -0
  4. data/README.md +126 -2
  5. data/VERSION +1 -1
  6. data/lib/patient_http/sidekiq/configuration.rb +258 -1
  7. data/lib/patient_http/sidekiq/direct_task_handler.rb +46 -0
  8. data/lib/patient_http/sidekiq/processor_observer.rb +162 -21
  9. data/lib/patient_http/sidekiq/redis_pool.rb +88 -0
  10. data/lib/patient_http/sidekiq/request_executor.rb +40 -10
  11. data/lib/patient_http/sidekiq/request_worker.rb +5 -2
  12. data/lib/patient_http/sidekiq/stats.rb +230 -33
  13. data/lib/patient_http/sidekiq/task_handler.rb +9 -3
  14. data/lib/patient_http/sidekiq/task_monitor.rb +410 -121
  15. data/lib/patient_http/sidekiq/task_monitor_thread.rb +41 -7
  16. data/lib/patient_http/sidekiq/web_ui/assets/patient-http/css/patient_http.css +29 -71
  17. data/lib/patient_http/sidekiq/web_ui/locales/ar.yml +10 -5
  18. data/lib/patient_http/sidekiq/web_ui/locales/cs.yml +10 -5
  19. data/lib/patient_http/sidekiq/web_ui/locales/da.yml +10 -5
  20. data/lib/patient_http/sidekiq/web_ui/locales/de.yml +10 -5
  21. data/lib/patient_http/sidekiq/web_ui/locales/el.yml +10 -5
  22. data/lib/patient_http/sidekiq/web_ui/locales/en.yml +10 -5
  23. data/lib/patient_http/sidekiq/web_ui/locales/es.yml +10 -5
  24. data/lib/patient_http/sidekiq/web_ui/locales/fa.yml +10 -5
  25. data/lib/patient_http/sidekiq/web_ui/locales/fr.yml +10 -5
  26. data/lib/patient_http/sidekiq/web_ui/locales/gd.yml +10 -5
  27. data/lib/patient_http/sidekiq/web_ui/locales/he.yml +10 -5
  28. data/lib/patient_http/sidekiq/web_ui/locales/hi.yml +10 -5
  29. data/lib/patient_http/sidekiq/web_ui/locales/it.yml +10 -5
  30. data/lib/patient_http/sidekiq/web_ui/locales/ja.yml +10 -5
  31. data/lib/patient_http/sidekiq/web_ui/locales/ko.yml +10 -5
  32. data/lib/patient_http/sidekiq/web_ui/locales/lt.yml +10 -5
  33. data/lib/patient_http/sidekiq/web_ui/locales/nb.yml +10 -5
  34. data/lib/patient_http/sidekiq/web_ui/locales/nl.yml +10 -5
  35. data/lib/patient_http/sidekiq/web_ui/locales/pl.yml +10 -5
  36. data/lib/patient_http/sidekiq/web_ui/locales/pt-BR.yml +10 -5
  37. data/lib/patient_http/sidekiq/web_ui/locales/pt.yml +10 -5
  38. data/lib/patient_http/sidekiq/web_ui/locales/ru.yml +10 -5
  39. data/lib/patient_http/sidekiq/web_ui/locales/sv.yml +10 -5
  40. data/lib/patient_http/sidekiq/web_ui/locales/ta.yml +10 -5
  41. data/lib/patient_http/sidekiq/web_ui/locales/tr.yml +10 -5
  42. data/lib/patient_http/sidekiq/web_ui/locales/uk.yml +10 -5
  43. data/lib/patient_http/sidekiq/web_ui/locales/ur.yml +10 -5
  44. data/lib/patient_http/sidekiq/web_ui/locales/vi.yml +10 -5
  45. data/lib/patient_http/sidekiq/web_ui/locales/zh-CN.yml +10 -5
  46. data/lib/patient_http/sidekiq/web_ui/locales/zh-TW.yml +10 -5
  47. data/lib/patient_http/sidekiq/web_ui/views/patient_http.html.erb +99 -44
  48. data/lib/patient_http/sidekiq/web_ui.rb +53 -1
  49. data/lib/patient_http/sidekiq.rb +303 -31
  50. data/patient_http-sidekiq.gemspec +1 -1
  51. metadata +6 -4
@@ -1,5 +1,8 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require "digest"
4
+ require "uri"
5
+
3
6
  module PatientHttp
4
7
  module Sidekiq
5
8
  # Manages inflight request tracking in Redis for crash recovery.
@@ -18,37 +21,79 @@ module PatientHttp
18
21
  # Redis key prefixes
19
22
  INFLIGHT_INDEX_KEY = "sidekiq:patient_http:inflight_index"
20
23
  INFLIGHT_JOBS_KEY = "sidekiq:patient_http:inflight_jobs"
24
+ INFLIGHT_DETAILS_KEY = "sidekiq:patient_http:inflight_details"
25
+ INFLIGHT_DETAILS_INDEX_KEY = "sidekiq:patient_http:inflight_details_index"
21
26
  PROCESS_SET_KEY = "sidekiq:patient_http:processes"
22
27
  GC_LOCK_KEY = "sidekiq:patient_http:gc_lock"
23
28
  GC_LAST_RUN_KEY = "sidekiq:patient_http:gc_last_run"
24
29
 
25
- # Lua script for atomic orphan removal.
26
- # Checks if the task is still orphaned (timestamp < threshold) and removes it atomically.
27
- # This prevents race conditions where a heartbeat could update the timestamp between
28
- # the check and the removal.
30
+ # Lua script for atomic orphan removal of a batch of request ids.
31
+ # For each id, checks that the task is still orphaned (timestamp <
32
+ # threshold) and removes it atomically, so a heartbeat cannot update the
33
+ # timestamp between the check and the removal. Ids that are no longer
34
+ # orphaned are skipped.
29
35
  #
30
36
  # KEYS[1] = index key (sorted set)
31
37
  # KEYS[2] = jobs key (hash)
32
- # ARGV[1] = request_id
33
- # ARGV[2] = threshold_ms
38
+ # KEYS[3] = details key (hash)
39
+ # KEYS[4] = details index key (sorted set)
40
+ # ARGV[1] = threshold_ms
41
+ # ARGV[2..] = request_ids
42
+ #
43
+ # Every removed id is returned, even when the jobs hash no longer holds
44
+ # its payload, so the caller can fall back to the payload it read before
45
+ # the script ran instead of losing the request.
34
46
  #
35
- # Returns: [removed (0/1), job_payload or nil]
47
+ # Returns: flat array of [request_id, job_payload, request_id, job_payload, ...]
48
+ # where job_payload is nil when the hash entry was already gone
36
49
  REMOVE_IF_ORPHANED_SCRIPT = <<~LUA
37
50
  local index_key = KEYS[1]
38
51
  local jobs_key = KEYS[2]
39
- local request_id = ARGV[1]
40
- local threshold_ms = tonumber(ARGV[2])
41
-
42
- local current_score = redis.call('ZSCORE', index_key, request_id)
43
- if not current_score or tonumber(current_score) >= threshold_ms then
44
- return {0, nil} -- Not orphaned or already removed
52
+ local details_key = KEYS[3]
53
+ local details_index_key = KEYS[4]
54
+ local threshold_ms = tonumber(ARGV[1])
55
+ local removed = {}
56
+
57
+ for i = 2, #ARGV do
58
+ local request_id = ARGV[i]
59
+ local current_score = redis.call('ZSCORE', index_key, request_id)
60
+ if current_score and tonumber(current_score) < threshold_ms then
61
+ local job_payload = redis.call('HGET', jobs_key, request_id)
62
+ redis.call('ZREM', index_key, request_id)
63
+ redis.call('HDEL', jobs_key, request_id)
64
+ redis.call('ZREM', details_index_key, request_id)
65
+ redis.call('HDEL', details_key, request_id)
66
+ table.insert(removed, request_id)
67
+ table.insert(removed, job_payload)
68
+ end
45
69
  end
46
70
 
47
- local job_payload = redis.call('HGET', jobs_key, request_id)
48
- redis.call('ZREM', index_key, request_id)
49
- redis.call('HDEL', jobs_key, request_id)
50
- return {1, job_payload}
71
+ return removed
51
72
  LUA
73
+ REMOVE_IF_ORPHANED_SHA = Digest::SHA1.hexdigest(REMOVE_IF_ORPHANED_SCRIPT).freeze
74
+
75
+ # Lua script for releasing the GC lock only when this process still owns
76
+ # it: a single-round-trip compare-and-delete.
77
+ #
78
+ # KEYS[1] = lock key
79
+ # ARGV[1] = lock identifier
80
+ #
81
+ # Returns: 1 if the lock was released, 0 otherwise
82
+ RELEASE_LOCK_SCRIPT = <<~LUA
83
+ if redis.call('GET', KEYS[1]) == ARGV[1] then
84
+ return redis.call('DEL', KEYS[1])
85
+ else
86
+ return 0
87
+ end
88
+ LUA
89
+ RELEASE_LOCK_SHA = Digest::SHA1.hexdigest(RELEASE_LOCK_SCRIPT).freeze
90
+
91
+ # Number of orphaned request ids processed per Lua call.
92
+ ORPHAN_BATCH_SIZE = 100
93
+
94
+ # Longest URL recorded for the Web UI, so that one enormous URL cannot
95
+ # take a disproportionate amount of memory.
96
+ MAX_DISPLAY_URL_LENGTH = 500
52
97
 
53
98
  # @return [Configuration] the configuration object
54
99
  attr_reader :config
@@ -65,10 +110,19 @@ module PatientHttp
65
110
 
66
111
  # Get all inflight counts across all processes and the number of max connections.
67
112
  #
68
- # @return [Hash] hash of "hostname:pid" => { inflight: Integer, max_capacity: Integer }
113
+ # The per-process inflight count comes from the shared inflight index, so
114
+ # it includes requests left behind by processes that have since died. The
115
+ # nested per-processor counts are snapshots each process publishes with
116
+ # its heartbeat, so they only cover processes that are still running and
117
+ # can lag by up to one monitor cycle.
118
+ #
119
+ # @return [Hash] hash of "hostname:pid" =>
120
+ # { inflight: Integer, max_capacity: Integer,
121
+ # processors: { String => { inflight: Integer, max_capacity: Integer } } }
69
122
  def inflight_counts_by_process
70
123
  process_ids = nil
71
124
  max_connections = nil
125
+ processor_snapshots = nil
72
126
  inflight_task_ids = nil
73
127
 
74
128
  ::Sidekiq.redis do |redis|
@@ -76,7 +130,10 @@ module PatientHttp
76
130
  return {} if process_ids.empty?
77
131
 
78
132
  max_keys = process_ids.map { |pid| max_connections_key_for(pid) }
79
- max_connections = redis.mget(*max_keys)
133
+ processor_keys = process_ids.map { |pid| processors_key_for(pid) }
134
+ values = redis.mget(*max_keys, *processor_keys)
135
+ max_connections = values.first(process_ids.size)
136
+ processor_snapshots = values.last(process_ids.size)
80
137
 
81
138
  inflight_task_ids = redis.zrange(INFLIGHT_INDEX_KEY, 0, -1)
82
139
  end
@@ -88,7 +145,7 @@ module PatientHttp
88
145
  result = {}
89
146
  stale_process_ids = []
90
147
 
91
- process_ids.zip(max_connections).each do |process_id, max_conn|
148
+ process_ids.zip(max_connections, processor_snapshots).each do |process_id, max_conn, snapshot|
92
149
  if max_conn.nil?
93
150
  # Mark for removal if max_conn key doesn't exist (process is gone)
94
151
  stale_process_ids << process_id
@@ -96,11 +153,12 @@ module PatientHttp
96
153
  host_pid = process_id.split(":", 3).first(2).join(":")
97
154
  counts = result[host_pid]
98
155
  unless counts
99
- counts = {inflight: 0, max_capacity: 0}
156
+ counts = {inflight: 0, max_capacity: 0, processors: {}}
100
157
  result[host_pid] = counts
101
158
  end
102
159
  counts[:inflight] += inflight_by_process_id[process_id]&.size.to_i
103
160
  counts[:max_capacity] += max_conn.to_i
161
+ merge_processor_snapshot(counts[:processors], snapshot)
104
162
  end
105
163
  end
106
164
 
@@ -114,6 +172,86 @@ module PatientHttp
114
172
  result
115
173
  end
116
174
 
175
+ # Get the inflight and capacity counts for each named processor across
176
+ # all running processes.
177
+ #
178
+ # @param processes [Hash, nil] the result of {inflight_counts_by_process};
179
+ # read from Redis when not given
180
+ # @return [Hash] hash of processor name => { inflight: Integer, max_capacity: Integer }
181
+ def inflight_counts_by_processor(processes = nil)
182
+ processes ||= inflight_counts_by_process
183
+
184
+ result = {}
185
+ processes.each_value do |data|
186
+ merge_processor_counts(result, data[:processors])
187
+ end
188
+ result.sort.to_h
189
+ end
190
+
191
+ # Get the details of the requests that have been in flight the longest.
192
+ #
193
+ # Only requests registered while +inflight_details+ was enabled are
194
+ # reported. A request stays listed while its crash-recovery record
195
+ # exists, so a request left behind by a process that died is listed
196
+ # until the orphan collector re-enqueues it.
197
+ #
198
+ # @param limit [Integer] maximum number of requests to return
199
+ # @return [Array<Hash>] oldest first, each with :request_id, :process_id,
200
+ # :url, :http_method, :processor, and :age in seconds
201
+ def inflight_details(limit: 50)
202
+ return [] if limit <= 0
203
+
204
+ task_ids = nil
205
+ timestamps = nil
206
+ records = nil
207
+
208
+ ::Sidekiq.redis do |redis|
209
+ entries = redis.zrange(INFLIGHT_DETAILS_INDEX_KEY, 0, limit - 1, withscores: true)
210
+ return [] if entries.empty?
211
+
212
+ task_ids = entries.map(&:first)
213
+ timestamps = entries.map(&:last)
214
+ records = redis.hmget(INFLIGHT_DETAILS_KEY, *task_ids)
215
+ end
216
+
217
+ now = Time.now.to_f
218
+ task_ids.zip(timestamps, records).filter_map do |task_id, timestamp_ms, record|
219
+ details = parse_details(record)
220
+ next unless details
221
+
222
+ process_id, request_id = task_id.split("/", 2)
223
+ {
224
+ request_id: request_id,
225
+ process_id: process_id.to_s.split(":", 3).first(2).join(":"),
226
+ url: details["url"],
227
+ http_method: details["method"],
228
+ processor: details["processor"],
229
+ age: (now - timestamp_ms.to_f / 1000.0).round(1)
230
+ }
231
+ end
232
+ end
233
+
234
+ # Remove the user name, password, query string, and fragment from a URL,
235
+ # keeping the scheme, host, and path. Used unless the configuration
236
+ # names its own sanitizer.
237
+ #
238
+ # @param url [String] the request URL
239
+ # @return [String] the URL to display
240
+ def sanitize_url(url)
241
+ uri = URI.parse(url.to_s)
242
+ uri.query = nil
243
+ uri.fragment = nil
244
+ # The password must be cleared before the user, and clearing the user
245
+ # info in one step does nothing.
246
+ uri.password = nil if uri.respond_to?(:password=)
247
+ uri.user = nil if uri.respond_to?(:user=)
248
+ uri.to_s
249
+ rescue
250
+ # A URL that cannot be parsed, such as one with a character outside
251
+ # US-ASCII, still must not carry credentials or a query string.
252
+ strip_credentials(url.to_s.split(/[?#]/, 2).first.to_s)
253
+ end
254
+
117
255
  # Get the total max connections across all processes
118
256
  #
119
257
  # @return [Integer] sum of max connections from all active processes
@@ -141,7 +279,10 @@ module PatientHttp
141
279
  end
142
280
 
143
281
  ::Sidekiq.redis do |redis|
144
- redis.del(INFLIGHT_INDEX_KEY, INFLIGHT_JOBS_KEY, PROCESS_SET_KEY, GC_LOCK_KEY, GC_LAST_RUN_KEY)
282
+ redis.del(
283
+ INFLIGHT_INDEX_KEY, INFLIGHT_JOBS_KEY, INFLIGHT_DETAILS_KEY,
284
+ INFLIGHT_DETAILS_INDEX_KEY, PROCESS_SET_KEY, GC_LOCK_KEY, GC_LAST_RUN_KEY
285
+ )
145
286
  end
146
287
  end
147
288
 
@@ -155,11 +296,94 @@ module PatientHttp
155
296
  def max_connections_key_for(process_id)
156
297
  "#{PROCESS_SET_KEY}:#{process_id}:max_connections"
157
298
  end
299
+
300
+ # Build the per-processor snapshot key for a given process identifier.
301
+ #
302
+ # @param process_id [String] the process identifier
303
+ #
304
+ # @return [String] the Redis key for the per-processor snapshot
305
+ def processors_key_for(process_id)
306
+ "#{PROCESS_SET_KEY}:#{process_id}:processors"
307
+ end
308
+
309
+ # Remove anything between the scheme and the host of a URL.
310
+ #
311
+ # @param url [String] the URL
312
+ # @return [String] the URL without credentials
313
+ def strip_credentials(url)
314
+ url.sub(%r{\A([a-zA-Z][a-zA-Z0-9+.-]*://)[^/@]*@}, '\\1')
315
+ end
316
+
317
+ # Parse one stored details record, ignoring one that cannot be read.
318
+ #
319
+ # @param record [String, nil] the serialized record
320
+ # @return [Hash, nil] the parsed record
321
+ def parse_details(record)
322
+ return nil if record.nil?
323
+
324
+ details = JSON.parse(record)
325
+ details.is_a?(Hash) ? details : nil
326
+ rescue JSON::ParserError
327
+ nil
328
+ end
329
+
330
+ # Merge one process's published snapshot into a set of per-processor counts.
331
+ #
332
+ # A snapshot written by a process running a different version of the gem
333
+ # may not be readable; it is skipped rather than failing the whole report.
334
+ #
335
+ # @param counts [Hash] per-processor counts to merge into
336
+ # @param snapshot [String, nil] the serialized snapshot
337
+ #
338
+ # @return [void]
339
+ def merge_processor_snapshot(counts, snapshot)
340
+ return if snapshot.nil?
341
+
342
+ parsed = begin
343
+ JSON.parse(snapshot)
344
+ rescue JSON::ParserError
345
+ return
346
+ end
347
+ return unless parsed.is_a?(Hash)
348
+
349
+ merge_processor_counts(
350
+ counts,
351
+ parsed.transform_values do |values|
352
+ next {} unless values.is_a?(Hash)
353
+
354
+ {inflight: values["inflight"].to_i, max_capacity: values["max_capacity"].to_i}
355
+ end
356
+ )
357
+ end
358
+
359
+ # Add per-processor counts into an accumulator.
360
+ #
361
+ # @param counts [Hash] per-processor counts to merge into
362
+ # @param additions [Hash, nil] per-processor counts to add
363
+ #
364
+ # @return [void]
365
+ def merge_processor_counts(counts, additions)
366
+ additions&.each do |name, values|
367
+ totals = (counts[name] ||= {inflight: 0, max_capacity: 0})
368
+ totals[:inflight] += values[:inflight].to_i
369
+ totals[:max_capacity] += values[:max_capacity].to_i
370
+ end
371
+ end
158
372
  end
159
373
 
160
374
  # @param config [Configuration] the configuration object
161
- def initialize(config)
375
+ # @param max_connections [#call, nil] callable returning the process's total
376
+ # configured max connections; defaults to the configuration's value. Ignored
377
+ # when a +processors+ source is given, which carries the same information
378
+ # per processor.
379
+ # @param processors [#call, nil] callable returning a snapshot of the
380
+ # process's processors as a hash of name => { inflight:, max_capacity: }.
381
+ # The snapshot is published with each heartbeat so the Web UI can report
382
+ # capacity per processor.
383
+ def initialize(config, max_connections: nil, processors: nil)
162
384
  @config = config
385
+ @max_connections_source = max_connections || -> { config.max_connections }
386
+ @processors_source = processors
163
387
  hostname = ::Socket.gethostname.force_encoding("UTF-8").tr(":/", "-")
164
388
  pid = ::Process.pid
165
389
  @lock_identifier = "#{hostname}:#{pid}:#{SecureRandom.hex(8)}".freeze
@@ -168,19 +392,28 @@ module PatientHttp
168
392
  # Register a request as inflight in Redis.
169
393
  #
170
394
  # @param task [RequestTask] the request task to register
395
+ # @param processor_name [Symbol, String, nil] name of the processor running
396
+ # the request, recorded with the request details
171
397
  #
172
398
  # @return [void]
173
- def register(task)
399
+ def register(task, processor_name: nil)
174
400
  timestamp_ms = (Time.now.to_f * 1000).round
175
401
  job_payload = JSON.generate(task.task_handler.sidekiq_job)
176
402
  task_id = full_task_id(task.id)
403
+ details = request_details(task, processor_name)
177
404
 
178
- ::Sidekiq.redis do |redis|
405
+ PatientHttp::Sidekiq.redis do |redis|
179
406
  redis.multi do |transaction|
180
407
  transaction.zadd(INFLIGHT_INDEX_KEY, timestamp_ms, task_id)
181
408
  transaction.hset(INFLIGHT_JOBS_KEY, task_id, job_payload)
182
409
  transaction.expire(INFLIGHT_INDEX_KEY, inflight_ttl)
183
410
  transaction.expire(INFLIGHT_JOBS_KEY, inflight_ttl)
411
+ if details
412
+ transaction.zadd(INFLIGHT_DETAILS_INDEX_KEY, timestamp_ms, task_id)
413
+ transaction.hset(INFLIGHT_DETAILS_KEY, task_id, details)
414
+ transaction.expire(INFLIGHT_DETAILS_INDEX_KEY, inflight_ttl)
415
+ transaction.expire(INFLIGHT_DETAILS_KEY, inflight_ttl)
416
+ end
184
417
  end
185
418
  end
186
419
  end
@@ -193,10 +426,12 @@ module PatientHttp
193
426
  def unregister(task)
194
427
  task_id = full_task_id(task.id)
195
428
 
196
- ::Sidekiq.redis do |redis|
429
+ PatientHttp::Sidekiq.redis do |redis|
197
430
  redis.multi do |transaction|
198
431
  transaction.zrem(INFLIGHT_INDEX_KEY, task_id)
199
432
  transaction.hdel(INFLIGHT_JOBS_KEY, task_id)
433
+ transaction.zrem(INFLIGHT_DETAILS_INDEX_KEY, task_id)
434
+ transaction.hdel(INFLIGHT_DETAILS_KEY, task_id)
200
435
  end
201
436
  end
202
437
  end
@@ -205,9 +440,12 @@ module PatientHttp
205
440
  #
206
441
  # @return [void]
207
442
  def remove_process
208
- ::Sidekiq.redis do |redis|
209
- redis.srem(PROCESS_SET_KEY, @lock_identifier)
210
- redis.del(max_connections_key)
443
+ PatientHttp::Sidekiq.redis do |redis|
444
+ redis.pipelined do |pipeline|
445
+ pipeline.srem(PROCESS_SET_KEY, @lock_identifier)
446
+ pipeline.del(max_connections_key)
447
+ pipeline.del(processors_key)
448
+ end
211
449
  end
212
450
  end
213
451
 
@@ -221,7 +459,7 @@ module PatientHttp
221
459
 
222
460
  timestamp_ms = (Time.now.to_f * 1000).round
223
461
 
224
- ::Sidekiq.redis do |redis|
462
+ PatientHttp::Sidekiq.redis do |redis|
225
463
  redis.pipelined do |pipeline|
226
464
  task_ids.each do |task_id|
227
465
  pipeline.call("ZADD", INFLIGHT_INDEX_KEY, "XX", timestamp_ms, full_task_id(task_id))
@@ -231,6 +469,8 @@ module PatientHttp
231
469
  # are registered.
232
470
  pipeline.call("EXPIRE", INFLIGHT_INDEX_KEY, inflight_ttl)
233
471
  pipeline.call("EXPIRE", INFLIGHT_JOBS_KEY, inflight_ttl)
472
+ pipeline.call("EXPIRE", INFLIGHT_DETAILS_INDEX_KEY, inflight_ttl)
473
+ pipeline.call("EXPIRE", INFLIGHT_DETAILS_KEY, inflight_ttl)
234
474
  end
235
475
  end
236
476
  end
@@ -242,7 +482,7 @@ module PatientHttp
242
482
  # @return [Boolean] true if registered, false otherwise
243
483
  # @api private
244
484
  def registered?(task)
245
- ::Sidekiq.redis do |redis|
485
+ PatientHttp::Sidekiq.redis do |redis|
246
486
  !redis.zscore(INFLIGHT_INDEX_KEY, full_task_id(task.id)).nil?
247
487
  end
248
488
  end
@@ -254,7 +494,7 @@ module PatientHttp
254
494
  # @return [Integer, nil] timestamp in milliseconds, or nil if not registered
255
495
  # @api private
256
496
  def heartbeat_timestamp_for(task)
257
- score = ::Sidekiq.redis do |redis|
497
+ score = PatientHttp::Sidekiq.redis do |redis|
258
498
  redis.zscore(INFLIGHT_INDEX_KEY, full_task_id(task.id))
259
499
  end
260
500
  score&.to_i
@@ -265,7 +505,7 @@ module PatientHttp
265
505
  # @return [Array<String>] list of full task IDs
266
506
  # @api private
267
507
  def registered_task_ids
268
- ::Sidekiq.redis do |redis|
508
+ PatientHttp::Sidekiq.redis do |redis|
269
509
  redis.zrange(INFLIGHT_INDEX_KEY, 0, -1)
270
510
  end.select { |id| id.start_with?("#{@lock_identifier}/") }
271
511
  end
@@ -278,18 +518,31 @@ module PatientHttp
278
518
  "#{@lock_identifier}/#{task_id}"
279
519
  end
280
520
 
281
- # Record the current process's max connections in Redis.
521
+ # Record the current process's capacity in Redis.
282
522
  #
283
- # This is used for monitoring purposes.
523
+ # This is used for monitoring purposes. The max connections key doubles as
524
+ # the process's liveness marker: it is refreshed on every heartbeat with a
525
+ # TTL shorter than the process set's, so a member of the set whose key has
526
+ # expired belongs to a process that is gone.
284
527
  #
285
528
  # @return [void]
286
529
  def ping_process
287
- ::Sidekiq.redis do |redis|
530
+ snapshot = @processors_source&.call
531
+ max_connections = if snapshot
532
+ snapshot.values.sum { |counts| counts[:max_capacity].to_i }
533
+ else
534
+ @max_connections_source.call
535
+ end
536
+
537
+ PatientHttp::Sidekiq.redis do |redis|
288
538
  redis.multi do |transaction|
289
539
  transaction.sadd(PROCESS_SET_KEY, @lock_identifier)
290
- transaction.set(max_connections_key, @config.max_connections)
540
+ transaction.set(max_connections_key, max_connections)
291
541
  transaction.expire(PROCESS_SET_KEY, inflight_ttl)
292
542
  transaction.expire(max_connections_key, process_ttl)
543
+ if snapshot
544
+ transaction.set(processors_key, serialize_processor_snapshot(snapshot), ex: process_ttl)
545
+ end
293
546
  end
294
547
  end
295
548
  end
@@ -298,41 +551,24 @@ module PatientHttp
298
551
  #
299
552
  # @return [Boolean] true if lock acquired, false otherwise
300
553
  def acquire_gc_lock
301
- ::Sidekiq.redis do |redis|
554
+ PatientHttp::Sidekiq.redis do |redis|
302
555
  # Use SET with NX and EX options directly
303
- # Returns "OK" if successful with ::Sidekiq.redis, nil if key already exists
556
+ # Returns "OK" if successful, nil if key already exists
304
557
  !!redis.set(GC_LOCK_KEY, @lock_identifier, nx: true, ex: gc_lock_ttl)
305
558
  end
306
559
  end
307
560
 
308
561
  # Release the garbage collection lock if held by this process.
309
562
  #
310
- # Uses Redis WATCH/MULTI/EXEC for optimistic locking to ensure we only
311
- # delete the lock if it's still held by this process.
563
+ # Uses a compare-and-delete Lua script so the check and deletion happen
564
+ # atomically in a single round trip.
312
565
  #
313
566
  # @return [Boolean] true if the lock was released, false otherwise
314
567
  def release_gc_lock
315
- ::Sidekiq.redis do |redis|
316
- # Watch the lock key for changes
317
- redis.watch(GC_LOCK_KEY)
318
-
319
- # Get current lock value
320
- current_value = redis.get(GC_LOCK_KEY)
321
-
322
- if current_value == @lock_identifier
323
- # Lock is ours, delete it atomically
324
- result = redis.multi do |transaction|
325
- transaction.del(GC_LOCK_KEY)
326
- end
327
- # MULTI returns nil if transaction was aborted (someone else modified the key)
328
- # Otherwise returns array with results
329
- !result.nil?
330
- else
331
- # Lock is not ours or doesn't exist
332
- redis.unwatch
333
- false
334
- end
568
+ result = PatientHttp::Sidekiq.redis do |redis|
569
+ run_script(redis, RELEASE_LOCK_SCRIPT, RELEASE_LOCK_SHA, [GC_LOCK_KEY], [@lock_identifier])
335
570
  end
571
+ result == 1
336
572
  end
337
573
 
338
574
  # Check if garbage collection should run based on the last run timestamp.
@@ -342,7 +578,7 @@ module PatientHttp
342
578
  #
343
579
  # @return [Boolean] true if GC should run, false otherwise
344
580
  def gc_needed?
345
- last_run = ::Sidekiq.redis do |redis|
581
+ last_run = PatientHttp::Sidekiq.redis do |redis|
346
582
  redis.get(GC_LAST_RUN_KEY)
347
583
  end
348
584
 
@@ -359,7 +595,7 @@ module PatientHttp
359
595
  #
360
596
  # @return [void]
361
597
  def record_gc_run
362
- ::Sidekiq.redis do |redis|
598
+ PatientHttp::Sidekiq.redis do |redis|
363
599
  redis.set(GC_LAST_RUN_KEY, (Time.now.to_f * 1000).floor, ex: gc_last_run_ttl)
364
600
  end
365
601
  end
@@ -397,7 +633,7 @@ module PatientHttp
397
633
  # @return [Array<Array(String, String)>] array of [request_id, job_payload] pairs
398
634
  def fetch_orphaned_requests(threshold_timestamp_ms)
399
635
  # Find all requests older than the threshold
400
- all_orphaned_request_ids = ::Sidekiq.redis do |redis|
636
+ all_orphaned_request_ids = PatientHttp::Sidekiq.redis do |redis|
401
637
  redis.zrange(INFLIGHT_INDEX_KEY, "-inf", threshold_timestamp_ms, byscore: true)
402
638
  end
403
639
 
@@ -412,7 +648,7 @@ module PatientHttp
412
648
  return [] if orphaned_request_ids.empty?
413
649
 
414
650
  # Retrieve job payloads for all orphaned requests
415
- job_payloads = ::Sidekiq.redis do |redis|
651
+ job_payloads = PatientHttp::Sidekiq.redis do |redis|
416
652
  redis.hmget(INFLIGHT_JOBS_KEY, *orphaned_request_ids)
417
653
  end
418
654
 
@@ -432,13 +668,13 @@ module PatientHttp
432
668
  #
433
669
  # @return [Array<String>] the subset of process IDs that are live
434
670
  def prune_stale_processes(process_ids)
435
- registered_ids = ::Sidekiq.redis do |redis|
671
+ registered_ids = PatientHttp::Sidekiq.redis do |redis|
436
672
  redis.smembers(PROCESS_SET_KEY)
437
673
  end
438
674
  candidates = process_ids & registered_ids
439
675
  return [] if candidates.empty?
440
676
 
441
- max_connection_values = ::Sidekiq.redis do |redis|
677
+ max_connection_values = PatientHttp::Sidekiq.redis do |redis|
442
678
  redis.mget(*candidates.map { |process_id| max_connections_key_for(process_id) })
443
679
  end
444
680
 
@@ -447,7 +683,7 @@ module PatientHttp
447
683
  .map { |pairs| pairs.map(&:first) }
448
684
 
449
685
  unless stale_process_ids.empty?
450
- ::Sidekiq.redis do |redis|
686
+ PatientHttp::Sidekiq.redis do |redis|
451
687
  redis.srem(PROCESS_SET_KEY, stale_process_ids)
452
688
  end
453
689
  end
@@ -457,6 +693,10 @@ module PatientHttp
457
693
 
458
694
  # Re-enqueue all orphaned jobs.
459
695
  #
696
+ # Ids are processed in batches: each batch is atomically checked and
697
+ # removed in one Lua call, then the removed jobs are pushed back to
698
+ # Sidekiq one by one (preserving each job's class, queue, and jid).
699
+ #
460
700
  # @param orphaned_requests [Array<Array(String, String)>] array of [request_id, job_payload] pairs
461
701
  # @param threshold_timestamp_ms [Integer] threshold timestamp in milliseconds
462
702
  # @param logger [Logger] logger for output
@@ -464,76 +704,106 @@ module PatientHttp
464
704
  # @return [Integer] number of jobs successfully re-enqueued
465
705
  def reenqueue_orphaned_jobs(orphaned_requests, threshold_timestamp_ms, logger)
466
706
  reenqueued_count = 0
467
-
468
- orphaned_requests.each do |request_id, job_payload|
469
- if reenqueue_orphaned_job(request_id, job_payload, threshold_timestamp_ms, logger)
470
- reenqueued_count += 1
707
+ # Payloads read before the script ran, used when the jobs hash entry
708
+ # was removed between the read and the script.
709
+ known_payloads = orphaned_requests.to_h
710
+
711
+ orphaned_requests.map(&:first).each_slice(ORPHAN_BATCH_SIZE) do |request_ids|
712
+ removed = remove_if_orphaned(request_ids, threshold_timestamp_ms)
713
+
714
+ removed.each_slice(2) do |request_id, job_payload|
715
+ job_payload ||= known_payloads[request_id]
716
+ next if job_payload.nil?
717
+
718
+ begin
719
+ job_hash = JSON.parse(job_payload)
720
+ ::Sidekiq::Client.push(job_hash)
721
+ reenqueued_count += 1
722
+
723
+ logger&.info(
724
+ "[PatientHttp::Sidekiq] Re-enqueued orphaned request #{request_id} to #{job_hash["class"]}"
725
+ )
726
+ rescue => e
727
+ logger&.error(
728
+ "[PatientHttp::Sidekiq] Failed to re-enqueue orphaned request #{request_id}: #{e.class} - #{e.message}"
729
+ )
730
+ end
471
731
  end
472
732
  end
473
733
 
474
734
  reenqueued_count
475
735
  end
476
736
 
477
- # Re-enqueue a single orphaned job using atomic Lua script.
737
+ # Atomically check a batch of ids and remove the ones still orphaned.
478
738
  #
479
- # This method atomically checks if the task is still orphaned and removes it
480
- # in a single Redis operation, preventing race conditions where a heartbeat
481
- # could update the timestamp between checking and removal.
739
+ # Uses a Lua script so the check and removal happen in a single atomic
740
+ # operation, preventing race conditions with heartbeat updates.
482
741
  #
483
- # @param request_id [String] the request ID
484
- # @param job_payload [String] the JSON job payload (used as fallback)
742
+ # @param request_ids [Array<String>] the request IDs to check
485
743
  # @param threshold_timestamp_ms [Integer] threshold timestamp in milliseconds
486
- # @param logger [Logger] logger for output
487
744
  #
488
- # @return [Boolean] true if successfully re-enqueued, false otherwise
489
- def reenqueue_orphaned_job(request_id, job_payload, threshold_timestamp_ms, logger)
490
- # Atomically check and remove if still orphaned
491
- removed, payload = remove_if_orphaned(request_id, threshold_timestamp_ms)
492
-
493
- return false unless removed == 1
494
-
495
- # Use payload from Lua script, fall back to provided payload
496
- actual_payload = payload || job_payload
497
- return false if actual_payload.nil?
498
-
499
- # Re-enqueue the job
500
- job_hash = JSON.parse(actual_payload)
501
- ::Sidekiq::Client.push(job_hash)
745
+ # @return [Array<String>] flat array of [request_id, job_payload, ...] pairs
746
+ def remove_if_orphaned(request_ids, threshold_timestamp_ms)
747
+ PatientHttp::Sidekiq.redis do |redis|
748
+ run_script(
749
+ redis,
750
+ REMOVE_IF_ORPHANED_SCRIPT,
751
+ REMOVE_IF_ORPHANED_SHA,
752
+ [INFLIGHT_INDEX_KEY, INFLIGHT_JOBS_KEY, INFLIGHT_DETAILS_KEY, INFLIGHT_DETAILS_INDEX_KEY],
753
+ [threshold_timestamp_ms.to_s, *request_ids]
754
+ )
755
+ end
756
+ end
502
757
 
503
- logger&.info(
504
- "[PatientHttp::Sidekiq] Re-enqueued orphaned request #{request_id} to #{job_hash["class"]}"
505
- )
758
+ # Run a Lua script by its SHA, falling back to a full EVAL (which also
759
+ # loads the script into the server's cache) when the server does not
760
+ # know the script yet.
761
+ #
762
+ # @param redis [Object] the Redis connection
763
+ # @param script [String] the Lua source
764
+ # @param sha [String] the precomputed SHA1 of the source
765
+ # @param keys [Array<String>] script KEYS
766
+ # @param argv [Array<String>] script ARGV
767
+ # @return [Object] the script's return value
768
+ def run_script(redis, script, sha, keys, argv)
769
+ redis.call("EVALSHA", sha, keys.size, *keys, *argv)
770
+ rescue RedisClient::CommandError => e
771
+ raise unless e.message.include?("NOSCRIPT")
772
+
773
+ redis.call("EVAL", script, keys.size, *keys, *argv)
774
+ end
506
775
 
507
- true
776
+ # Build the serialized details recorded for a request, or nil when the
777
+ # details are turned off or cannot be built. A failure here must not stop
778
+ # the request from being registered, so it is logged and skipped.
779
+ #
780
+ # @param task [RequestTask] the request task
781
+ # @param processor_name [Symbol, String, nil] the processor running the request
782
+ # @return [String, nil] the serialized details
783
+ def request_details(task, processor_name)
784
+ return nil unless config.inflight_details?
785
+
786
+ request = task.request
787
+ JSON.generate({
788
+ "url" => display_url(request.url),
789
+ "method" => request.http_method.to_s,
790
+ "processor" => processor_name&.to_s
791
+ }.compact)
508
792
  rescue => e
509
- logger&.error(
510
- "[PatientHttp::Sidekiq] Failed to re-enqueue orphaned request #{request_id}: #{e.class} - #{e.message}"
793
+ config.logger&.warn(
794
+ "[PatientHttp::Sidekiq] Failed to record the details of request #{task.id}: #{e.class} - #{e.message}"
511
795
  )
512
- false
796
+ nil
513
797
  end
514
798
 
515
- # Atomically check if orphaned and remove from registry.
799
+ # The URL to record for a request, sanitized and bounded in length.
516
800
  #
517
- # Uses a Lua script to ensure the check and removal happen in a single
518
- # atomic operation, preventing race conditions with heartbeat updates.
519
- #
520
- # @param request_id [String] the request ID
521
- # @param threshold_timestamp_ms [Integer] threshold timestamp in milliseconds
522
- #
523
- # @return [Array(Integer, String)] [removed (0/1), job_payload or nil]
524
- def remove_if_orphaned(request_id, threshold_timestamp_ms)
525
- ::Sidekiq.redis do |redis|
526
- # EVAL script numkeys key1 key2 arg1 arg2
527
- redis.call(
528
- "EVAL",
529
- REMOVE_IF_ORPHANED_SCRIPT,
530
- 2, # number of keys
531
- INFLIGHT_INDEX_KEY,
532
- INFLIGHT_JOBS_KEY,
533
- request_id,
534
- threshold_timestamp_ms.to_s
535
- )
536
- end
801
+ # @param url [String] the request URL
802
+ # @return [String] the URL to display
803
+ def display_url(url)
804
+ sanitizer = config.inflight_url_sanitizer
805
+ sanitized = sanitizer ? sanitizer.call(url) : self.class.sanitize_url(url)
806
+ sanitized.to_s[0, MAX_DISPLAY_URL_LENGTH]
537
807
  end
538
808
 
539
809
  # Calculate the TTL for inflight data structures.
@@ -580,6 +850,25 @@ module PatientHttp
580
850
  def max_connections_key_for(process_id)
581
851
  "#{PROCESS_SET_KEY}:#{process_id}:max_connections"
582
852
  end
853
+
854
+ def processors_key
855
+ "#{PROCESS_SET_KEY}:#{@lock_identifier}:processors"
856
+ end
857
+
858
+ # Serialize a per-processor snapshot for publication.
859
+ #
860
+ # @param snapshot [Hash] processor name => { inflight:, max_capacity: }
861
+ # @return [String] the serialized snapshot
862
+ def serialize_processor_snapshot(snapshot)
863
+ JSON.generate(
864
+ snapshot.each_with_object({}) do |(name, counts), hash|
865
+ hash[name.to_s] = {
866
+ "inflight" => counts[:inflight].to_i,
867
+ "max_capacity" => counts[:max_capacity].to_i
868
+ }
869
+ end
870
+ )
871
+ end
583
872
  end
584
873
  end
585
874
  end