patient_http-sidekiq 1.2.0 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. checksums.yaml +4 -4
  2. data/ARCHITECTURE.md +31 -10
  3. data/CHANGELOG.md +32 -0
  4. data/README.md +126 -2
  5. data/VERSION +1 -1
  6. data/lib/patient_http/sidekiq/configuration.rb +258 -1
  7. data/lib/patient_http/sidekiq/direct_task_handler.rb +46 -0
  8. data/lib/patient_http/sidekiq/processor_observer.rb +162 -21
  9. data/lib/patient_http/sidekiq/redis_pool.rb +88 -0
  10. data/lib/patient_http/sidekiq/request_executor.rb +40 -10
  11. data/lib/patient_http/sidekiq/request_worker.rb +5 -2
  12. data/lib/patient_http/sidekiq/stats.rb +230 -33
  13. data/lib/patient_http/sidekiq/task_handler.rb +9 -3
  14. data/lib/patient_http/sidekiq/task_monitor.rb +410 -121
  15. data/lib/patient_http/sidekiq/task_monitor_thread.rb +41 -7
  16. data/lib/patient_http/sidekiq/web_ui/assets/patient-http/css/patient_http.css +29 -71
  17. data/lib/patient_http/sidekiq/web_ui/locales/ar.yml +10 -5
  18. data/lib/patient_http/sidekiq/web_ui/locales/cs.yml +10 -5
  19. data/lib/patient_http/sidekiq/web_ui/locales/da.yml +10 -5
  20. data/lib/patient_http/sidekiq/web_ui/locales/de.yml +10 -5
  21. data/lib/patient_http/sidekiq/web_ui/locales/el.yml +10 -5
  22. data/lib/patient_http/sidekiq/web_ui/locales/en.yml +10 -5
  23. data/lib/patient_http/sidekiq/web_ui/locales/es.yml +10 -5
  24. data/lib/patient_http/sidekiq/web_ui/locales/fa.yml +10 -5
  25. data/lib/patient_http/sidekiq/web_ui/locales/fr.yml +10 -5
  26. data/lib/patient_http/sidekiq/web_ui/locales/gd.yml +10 -5
  27. data/lib/patient_http/sidekiq/web_ui/locales/he.yml +10 -5
  28. data/lib/patient_http/sidekiq/web_ui/locales/hi.yml +10 -5
  29. data/lib/patient_http/sidekiq/web_ui/locales/it.yml +10 -5
  30. data/lib/patient_http/sidekiq/web_ui/locales/ja.yml +10 -5
  31. data/lib/patient_http/sidekiq/web_ui/locales/ko.yml +10 -5
  32. data/lib/patient_http/sidekiq/web_ui/locales/lt.yml +10 -5
  33. data/lib/patient_http/sidekiq/web_ui/locales/nb.yml +10 -5
  34. data/lib/patient_http/sidekiq/web_ui/locales/nl.yml +10 -5
  35. data/lib/patient_http/sidekiq/web_ui/locales/pl.yml +10 -5
  36. data/lib/patient_http/sidekiq/web_ui/locales/pt-BR.yml +10 -5
  37. data/lib/patient_http/sidekiq/web_ui/locales/pt.yml +10 -5
  38. data/lib/patient_http/sidekiq/web_ui/locales/ru.yml +10 -5
  39. data/lib/patient_http/sidekiq/web_ui/locales/sv.yml +10 -5
  40. data/lib/patient_http/sidekiq/web_ui/locales/ta.yml +10 -5
  41. data/lib/patient_http/sidekiq/web_ui/locales/tr.yml +10 -5
  42. data/lib/patient_http/sidekiq/web_ui/locales/uk.yml +10 -5
  43. data/lib/patient_http/sidekiq/web_ui/locales/ur.yml +10 -5
  44. data/lib/patient_http/sidekiq/web_ui/locales/vi.yml +10 -5
  45. data/lib/patient_http/sidekiq/web_ui/locales/zh-CN.yml +10 -5
  46. data/lib/patient_http/sidekiq/web_ui/locales/zh-TW.yml +10 -5
  47. data/lib/patient_http/sidekiq/web_ui/views/patient_http.html.erb +99 -44
  48. data/lib/patient_http/sidekiq/web_ui.rb +53 -1
  49. data/lib/patient_http/sidekiq.rb +303 -31
  50. data/patient_http-sidekiq.gemspec +1 -1
  51. metadata +6 -4
@@ -2,47 +2,188 @@
2
2
 
3
3
  module PatientHttp
4
4
  module Sidekiq
5
- # Procesor Observer that collect stats in Redis for the WebUI and
6
- # monitors for crashed processes in order to re-enqueue workers.
5
+ # Processor observer that records stats and maintains the crash-recovery
6
+ # registry for one processor. The stats aggregator and task monitor are
7
+ # shared across all processors in the process; the module owns them and
8
+ # the monitor thread.
9
+ #
10
+ # Tasks are registered in the crash-recovery registry when the processor
11
+ # accepts them, before Processor#enqueue returns, so a request always has
12
+ # a durable record from the moment the caller hands it off. The entry is
13
+ # removed when the request completes or when a Sidekiq job owns the
14
+ # request again (the task was rejected or re-enqueued). When result
15
+ # delivery fails (completion_failed), the entry is kept so the orphan
16
+ # collector re-enqueues the request instead of losing it, unless the
17
+ # failure is one that trying again cannot fix; see
18
+ # +UNDELIVERABLE_RESULT_ERRORS+.
7
19
  class ProcessorObserver < PatientHttp::ProcessorObserver
20
+ # Delivery failures that mean the result can never be delivered: the
21
+ # payload cannot be serialized, so every re-enqueue would end the same
22
+ # way. A request that fails with one of these is moved to the Sidekiq
23
+ # dead set instead of being kept for crash recovery. Every other failure
24
+ # is treated as temporary (Redis unavailable, for example) and keeps its
25
+ # crash-recovery record.
26
+ UNDELIVERABLE_RESULT_ERRORS = [
27
+ JSON::GeneratorError,
28
+ Encoding::UndefinedConversionError,
29
+ Encoding::InvalidByteSequenceError,
30
+ Encoding::CompatibilityError
31
+ ].freeze
32
+
8
33
  attr_reader :task_monitor
9
34
 
10
- def initialize(processor)
35
+ def initialize(processor, stats:, task_monitor:)
11
36
  @processor = processor
12
- @stats = Stats.new(processor.config)
13
- @task_monitor = TaskMonitor.new(processor.config)
14
- @monitor_thread = TaskMonitorThread.new(
15
- processor.config,
16
- @task_monitor,
17
- -> { @processor.inflight_request_ids }
18
- )
37
+ @stats = stats
38
+ @task_monitor = task_monitor
39
+ @processor_name = processor.name
40
+ @requeued_task_ids = Set.new
41
+ @requeued_mutex = Mutex.new
19
42
  end
20
43
 
21
- def start
22
- @monitor_thread.start
44
+ def capacity_exceeded
45
+ @stats.record_capacity_exceeded(processor_name: @processor_name)
23
46
  end
24
47
 
25
- def stop
26
- @monitor_thread.stop
27
- task_monitor.remove_process
48
+ def request_enqueued(request_task)
49
+ task_monitor.register(request_task, processor_name: @processor_name)
50
+ @stats.record_inflight_peak(inflight_after_enqueue, processor_name: @processor_name)
28
51
  end
29
52
 
30
- def capacity_exceeded
31
- @stats.record_capacity_exceeded
53
+ def request_rejected(request_task)
54
+ task_monitor.unregister(request_task)
32
55
  end
33
56
 
34
- def request_start(request_task)
35
- task_monitor.register(request_task)
57
+ def request_requeued(request_task)
58
+ task_monitor.unregister(request_task)
59
+ # The re-enqueue path fires request_end after request_requeued, but
60
+ # only for tasks that already started. Remember those tasks so that
61
+ # request_end does not unregister a second time or record a completion
62
+ # stat for a request that never completed. A task that never started
63
+ # gets no request_end, so remembering it would leak the id forever.
64
+ return unless request_task.started?
65
+
66
+ @requeued_mutex.synchronize { @requeued_task_ids << request_task.id }
36
67
  end
37
68
 
38
69
  def request_end(request_task)
70
+ requeued = @requeued_mutex.synchronize { @requeued_task_ids.delete?(request_task.id) }
71
+ return if requeued
72
+
39
73
  task_monitor.unregister(request_task)
40
- @stats.record_request(request_task.response&.status, request_task.duration)
74
+ @stats.record_request(request_task.response&.status, request_task.duration, processor_name: @processor_name)
41
75
  end
42
76
 
43
77
  def request_error(error)
44
78
  error_type = error.is_a?(PatientHttp::Error) ? error.error_type : :exception
45
- @stats.record_error(error_type)
79
+ @stats.record_error(error_type, processor_name: @processor_name)
80
+ end
81
+
82
+ def completion_failed(request_task, error)
83
+ if undeliverable_result?(error) && kill_job(request_task, error)
84
+ # The request itself finished and nothing will deliver its result,
85
+ # so record it and drop its crash-recovery entry.
86
+ task_monitor.unregister(request_task)
87
+ @stats.record_request(request_task.response&.status, request_task.duration, processor_name: @processor_name)
88
+ @stats.record_error(:undeliverable_result, processor_name: @processor_name)
89
+ PatientHttp::Sidekiq.configuration.logger&.error(
90
+ "[PatientHttp::Sidekiq] Result for request #{request_task.id} can never be delivered; " \
91
+ "moved its job to the dead set: #{error.class} - #{error_message(error)}"
92
+ )
93
+ return
94
+ end
95
+
96
+ # Keep the crash-recovery registry entry: the orphan collector will
97
+ # re-enqueue the request once its heartbeat goes stale.
98
+ @stats.record_error(:completion_failed, processor_name: @processor_name)
99
+ PatientHttp::Sidekiq.configuration.logger&.error(
100
+ "[PatientHttp::Sidekiq] Result delivery failed for request #{request_task.id}; " \
101
+ "leaving crash-recovery record for re-enqueue: #{error.class} - #{error_message(error)}"
102
+ )
103
+ end
104
+
105
+ private
106
+
107
+ # The number of requests the processor holds once it accepts the request
108
+ # being announced. The count only rises when a request is accepted, so
109
+ # sampling it here catches every high-water mark.
110
+ #
111
+ # The announcement is made before the task is counted, so the task itself
112
+ # is added. A task announced when the processor is already full is
113
+ # rejected right after, so the count is held to the processor's capacity.
114
+ #
115
+ # @return [Integer]
116
+ def inflight_after_enqueue
117
+ [@processor.total_count + 1, @processor.config.max_connections].min
118
+ end
119
+
120
+ # Whether an error means the result can never be delivered. The cause
121
+ # chain is examined as well, because the failure is usually raised while
122
+ # the result is being written to Redis.
123
+ #
124
+ # @param error [Exception] the delivery failure
125
+ # @return [Boolean]
126
+ def undeliverable_result?(error)
127
+ while error
128
+ return true if UNDELIVERABLE_RESULT_ERRORS.any? { |error_class| error.is_a?(error_class) }
129
+
130
+ error = error.cause
131
+ end
132
+
133
+ false
134
+ end
135
+
136
+ # Move a request's job to the Sidekiq dead set, where it can be
137
+ # inspected and retried by hand.
138
+ #
139
+ # Sidekiq's API is loaded on demand because loading it eagerly fails on
140
+ # some supported Sidekiq versions. A job that cannot be moved reports
141
+ # false, so the caller keeps the crash-recovery record rather than
142
+ # dropping the request.
143
+ #
144
+ # @param request_task [RequestTask] the request task
145
+ # @param error [Exception] the delivery failure
146
+ # @return [Boolean] whether the job was moved
147
+ def kill_job(request_task, error)
148
+ require "sidekiq/api"
149
+
150
+ job = dead_job(request_task.task_handler.sidekiq_job, error)
151
+ ::Sidekiq::DeadSet.new.kill(JSON.generate(job), notify_failure: true, ex: error)
152
+ true
153
+ rescue LoadError, StandardError => e
154
+ PatientHttp::Sidekiq.configuration.logger&.error(
155
+ "[PatientHttp::Sidekiq] Failed to move request #{request_task.id} to the dead set: #{e.class} - #{error_message(e)}"
156
+ )
157
+ false
158
+ end
159
+
160
+ # Build the job record for the dead set. The failure fields are the ones
161
+ # Sidekiq writes when a job dies on its own, so the entry reads the same
162
+ # in the Web UI. A directly executed request has no job id yet, so it is
163
+ # given one; the Web UI identifies dead entries by it.
164
+ #
165
+ # @param job [Hash] the Sidekiq job hash
166
+ # @param error [Exception] the delivery failure
167
+ # @return [Hash] the job record to store
168
+ def dead_job(job, error)
169
+ job.merge(
170
+ "jid" => job["jid"] || SecureRandom.hex(12),
171
+ "failed_at" => Time.now.to_f,
172
+ "error_class" => error.class.name,
173
+ "error_message" => error_message(error)
174
+ )
175
+ end
176
+
177
+ # An error message that is safe to serialize and to log. A message that
178
+ # reports a byte the result could not be serialized with holds that byte
179
+ # itself, which would fail the same way the result did.
180
+ #
181
+ # @param error [Exception] the error
182
+ # @return [String] the message with any invalid bytes replaced
183
+ def error_message(error)
184
+ message = error.message.to_s
185
+ message = message.dup.force_encoding(Encoding::UTF_8) if message.encoding == Encoding::BINARY
186
+ message.scrub
46
187
  end
47
188
  end
48
189
  end
@@ -0,0 +1,88 @@
1
+ # frozen_string_literal: true
2
+
3
+ module PatientHttp
4
+ module Sidekiq
5
+ # Dedicated Redis connection pool for the gem's own threads.
6
+ #
7
+ # The processor's completion worker threads and the task monitor thread
8
+ # carry no Sidekiq capsule state, so plain `Sidekiq.redis` calls from them
9
+ # fall through to Sidekiq's small internal pool (10 connections, 1 second
10
+ # checkout timeout). Under load that pool becomes a serialization point
11
+ # and checkout timeouts can lose work. This pool is built from the
12
+ # application's own Sidekiq Redis configuration and is used for all
13
+ # registry, stats, and job pushes made from gem-owned threads.
14
+ class RedisPool
15
+ DEFAULT_MINIMUM_SIZE = 10
16
+
17
+ # @param config [Configuration] the gem configuration
18
+ def initialize(config)
19
+ @config = config
20
+ @pid = nil
21
+ @pool = nil
22
+ @mutex = Mutex.new
23
+ end
24
+
25
+ # The underlying ConnectionPool, created lazily and rebuilt after a
26
+ # process fork so child processes never share parent connections.
27
+ #
28
+ # @return [ConnectionPool]
29
+ def pool
30
+ @mutex.synchronize do
31
+ if @pool.nil? || @pid != ::Process.pid
32
+ @pool = ::Sidekiq.default_configuration.new_redis_pool(size, "patient_http")
33
+ @pid = ::Process.pid
34
+ end
35
+ @pool
36
+ end
37
+ end
38
+
39
+ # Check out a connection with the gem's checkout timeout and yield it.
40
+ # A connection-level failure is retried once on a fresh checkout,
41
+ # mirroring the retry Sidekiq itself performs.
42
+ #
43
+ # The retry replays the whole block, so callers whose block is not
44
+ # idempotent (counter increments, anything the server may already have
45
+ # applied before the connection dropped) must pass
46
+ # +retry_on_connection_error: false+ and handle the failure themselves.
47
+ #
48
+ # @param retry_on_connection_error [Boolean] whether to replay the block
49
+ # once after a connection-level failure
50
+ # @yield [conn] the Redis connection
51
+ # @return [Object] the block's return value
52
+ def with(retry_on_connection_error: true, &block)
53
+ retryable = retry_on_connection_error
54
+ begin
55
+ pool.with(timeout: @config.redis_pool_timeout) do |conn|
56
+ yield conn
57
+ end
58
+ rescue RedisClient::ConnectionError
59
+ raise unless retryable
60
+ retryable = false
61
+ retry
62
+ end
63
+ end
64
+
65
+ # Close all connections and drop the pool.
66
+ #
67
+ # @return [void]
68
+ def shutdown
69
+ @mutex.synchronize do
70
+ @pool&.shutdown { |conn| conn.close }
71
+ @pool = nil
72
+ @pid = nil
73
+ end
74
+ end
75
+
76
+ private
77
+
78
+ # Pool size: explicit configuration wins; otherwise size it to cover the
79
+ # completion worker threads plus the monitor thread and request
80
+ # registration, with a sane floor.
81
+ #
82
+ # @return [Integer]
83
+ def size
84
+ @config.redis_pool_size || [DEFAULT_MINIMUM_SIZE, @config.completion_threads + 3].max
85
+ end
86
+ end
87
+ end
88
+ end
@@ -7,9 +7,8 @@ module PatientHttp
7
7
  class << self
8
8
  # Execute the request directly on the async processor.
9
9
  #
10
- # This method enqueues the request directly to the async processor. It must be
11
- # called from within a Sidekiq job context (the sidekiq_job parameter is required).
12
- # Used internally by RequestWorker.
10
+ # This method enqueues the request directly to the async processor. Used internally
11
+ # by RequestWorker.
13
12
  #
14
13
  # When the request completes, the callback's +on_complete+ method is called with
15
14
  # a Response object. If an error occurs (network error, timeout, or non-2xx response
@@ -22,6 +21,9 @@ module PatientHttp
22
21
  # If not provided, uses PatientHttp::Sidekiq::Context.current_job.
23
22
  # This requires the PatientHttp::Sidekiq::Context::Middleware to be added
24
23
  # to the Sidekiq server middleware chain.
24
+ # @param task_handler [PatientHttp::TaskHandler, nil] A prebuilt task handler.
25
+ # When provided, the sidekiq_job parameter is ignored and no handler is
26
+ # built from it. Used for direct execution on the local processor.
25
27
  # @param synchronous [Boolean] If true, runs the request inline (for testing).
26
28
  # @param callback_args [#to_h, nil] Arguments to pass to callback via the
27
29
  # Response/Error object. Must respond to +to_h+ and contain only JSON-native types
@@ -32,20 +34,31 @@ module PatientHttp
32
34
  # and calls +on_error+ instead of +on_complete+. Defaults to false.
33
35
  # @param request_id [String, nil] Unique request ID for tracking. If nil, a new UUID
34
36
  # will be generated.
37
+ # @param processor_name [Symbol, String, nil] Name of the processor profile to run
38
+ # the request on. Defaults to the request's own processor name or :default.
35
39
  # @return [String] the request ID
36
40
  # @api private
37
41
  def execute(
38
42
  request,
39
43
  callback:,
40
44
  sidekiq_job: nil,
45
+ task_handler: nil,
41
46
  synchronous: false,
42
47
  callback_args: nil,
43
48
  raise_error_responses: false,
44
- request_id: nil
49
+ request_id: nil,
50
+ processor_name: nil
45
51
  )
46
- sidekiq_job = validate_sidekiq_job(sidekiq_job)
52
+ task_handler ||= TaskHandler.new(validate_sidekiq_job(sidekiq_job))
47
53
  config = PatientHttp::Sidekiq.configuration
48
- task_handler = TaskHandler.new(sidekiq_job)
54
+
55
+ # Look up the named processor and the effective configuration for its
56
+ # profile, so per-processor overrides apply to the request itself.
57
+ # A running processor already holds the built profile configuration.
58
+ name = (processor_name || request.processor || :default).to_sym
59
+ processor = PatientHttp::Sidekiq.processor(name)
60
+ profile_declared = config.processor_profiles.key?(name)
61
+ task_config = processor&.config || (profile_declared ? config.processor_config(name) : config)
49
62
 
50
63
  task = PatientHttp::RequestTask.new(
51
64
  request: request,
@@ -54,26 +67,43 @@ module PatientHttp
54
67
  callback_args: callback_args,
55
68
  raise_error_responses: raise_error_responses,
56
69
  id: request_id,
57
- default_max_redirects: config.max_redirects
70
+ default_max_redirects: task_config.max_redirects
58
71
  )
59
72
 
60
73
  # Run the request inline if Sidekiq::Testing.inline! is enabled
61
74
  if synchronous || async_disabled?
62
75
  PatientHttp::SynchronousExecutor.new(
63
76
  task,
64
- config: config,
77
+ config: task_config,
65
78
  on_complete: ->(response) { PatientHttp::Sidekiq.invoke_completion_callbacks(response) },
66
79
  on_error: ->(error) { PatientHttp::Sidekiq.invoke_error_callbacks(error) }
67
80
  ).call
68
81
  return task.id
69
82
  end
70
83
 
71
- # Check if processor is running
72
- processor = PatientHttp::Sidekiq.processor
84
+ # An unknown name raises so the job lands in Sidekiq's retry
85
+ # mechanism instead of being dropped; this covers rolling deploys
86
+ # where an old process has not configured a new profile yet.
87
+ if processor.nil? && !profile_declared
88
+ raise PatientHttp::UnknownProcessorError.new("No processor profile configured for #{name.inspect}")
89
+ end
90
+
73
91
  unless processor&.running?
74
92
  raise PatientHttp::NotRunningError.new("Cannot enqueue request: processor is not running")
75
93
  end
76
94
 
95
+ # Advisory capacity check before enqueueing. A real enqueue pays for
96
+ # durable registration before the authoritative capacity check, so a
97
+ # full processor would cost several Redis round trips just to be
98
+ # rejected. This peek rejects for free; the race where capacity fills
99
+ # after the peek falls through to the normal rejection path.
100
+ unless processor.capacity_available?
101
+ PatientHttp::Sidekiq.stats.record_capacity_exceeded(processor_name: name)
102
+ raise PatientHttp::MaxCapacityError.new(
103
+ "Cannot enqueue request: processor #{name} is at max capacity (#{processor.config.max_connections} connections)"
104
+ )
105
+ end
106
+
77
107
  processor.enqueue(task)
78
108
 
79
109
  task.id
@@ -40,8 +40,10 @@ module PatientHttp
40
40
  # nil is treated as false
41
41
  # @param callback_args [Hash, nil] Arguments to pass to the callback
42
42
  # @param request_id [String, nil] Unique request ID for tracking
43
+ # @param processor_name [String, nil] Name of the processor profile to run the request
44
+ # on; nil (jobs enqueued by older versions) runs on the default processor
43
45
  # @return [void]
44
- def perform(data, callback_service_name, raise_error_responses, callback_args, request_id)
46
+ def perform(data, callback_service_name, raise_error_responses, callback_args, request_id, processor_name = nil)
45
47
  # Fetch from external storage if needed
46
48
  actual_data = PatientHttp::ExternalStorage.storage_ref?(data) ? Sidekiq.external_storage.fetch(data) : data
47
49
  actual_data = Sidekiq.decrypt(actual_data)
@@ -59,7 +61,8 @@ module PatientHttp
59
61
  raise_error_responses: raise_error_responses,
60
62
  callback_args: callback_args,
61
63
  sidekiq_job: sidekiq_job,
62
- request_id: request_id
64
+ request_id: request_id,
65
+ processor_name: processor_name || "default"
63
66
  )
64
67
  end
65
68
  end