patient_http-sidekiq 1.2.0 → 1.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/ARCHITECTURE.md +31 -10
- data/CHANGELOG.md +32 -0
- data/README.md +126 -2
- data/VERSION +1 -1
- data/lib/patient_http/sidekiq/configuration.rb +258 -1
- data/lib/patient_http/sidekiq/direct_task_handler.rb +46 -0
- data/lib/patient_http/sidekiq/processor_observer.rb +162 -21
- data/lib/patient_http/sidekiq/redis_pool.rb +88 -0
- data/lib/patient_http/sidekiq/request_executor.rb +40 -10
- data/lib/patient_http/sidekiq/request_worker.rb +5 -2
- data/lib/patient_http/sidekiq/stats.rb +230 -33
- data/lib/patient_http/sidekiq/task_handler.rb +9 -3
- data/lib/patient_http/sidekiq/task_monitor.rb +410 -121
- data/lib/patient_http/sidekiq/task_monitor_thread.rb +41 -7
- data/lib/patient_http/sidekiq/web_ui/assets/patient-http/css/patient_http.css +29 -71
- data/lib/patient_http/sidekiq/web_ui/locales/ar.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/cs.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/da.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/de.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/el.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/en.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/es.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/fa.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/fr.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/gd.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/he.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/hi.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/it.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/ja.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/ko.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/lt.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/nb.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/nl.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/pl.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/pt-BR.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/pt.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/ru.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/sv.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/ta.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/tr.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/uk.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/ur.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/vi.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/zh-CN.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/locales/zh-TW.yml +10 -5
- data/lib/patient_http/sidekiq/web_ui/views/patient_http.html.erb +99 -44
- data/lib/patient_http/sidekiq/web_ui.rb +53 -1
- data/lib/patient_http/sidekiq.rb +303 -31
- data/patient_http-sidekiq.gemspec +1 -1
- metadata +6 -4
|
@@ -1,83 +1,212 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
|
+
require "digest"
|
|
4
|
+
|
|
3
5
|
module PatientHttp
|
|
4
6
|
module Sidekiq
|
|
5
|
-
#
|
|
7
|
+
# Tracks processor statistics with local aggregation.
|
|
6
8
|
#
|
|
7
|
-
#
|
|
8
|
-
#
|
|
9
|
-
#
|
|
9
|
+
# Metrics are accumulated in memory and flushed to a Redis hash on an
|
|
10
|
+
# interval (see +stats_flush_interval+), so recording a request costs a
|
|
11
|
+
# hash increment instead of a Redis round trip and the shared totals key
|
|
12
|
+
# is not a per-request hot key across the fleet. Deltas are commutative
|
|
13
|
+
# increments, so concurrent flushes from many processes are safe. A flush
|
|
14
|
+
# failure merges the deltas back so they are retried on the next flush;
|
|
15
|
+
# a crashed process loses at most one interval of counters.
|
|
10
16
|
class Stats
|
|
17
|
+
include PatientHttp::TimeHelper
|
|
18
|
+
|
|
11
19
|
# Redis key prefixes
|
|
12
20
|
TOTALS_KEY = "sidekiq:patient_http:totals"
|
|
13
21
|
|
|
14
22
|
# TTLs
|
|
15
23
|
TOTALS_TTL = 30 * 24 * 60 * 60 # 30 days in seconds
|
|
16
24
|
|
|
25
|
+
# Metrics reported for every processor, so that a processor that has only
|
|
26
|
+
# recorded some of them still reports the rest as zero.
|
|
27
|
+
PROCESSOR_METRICS = {
|
|
28
|
+
"requests" => 0,
|
|
29
|
+
"duration" => 0.0,
|
|
30
|
+
"errors" => 0,
|
|
31
|
+
"max_capacity_exceeded" => 0,
|
|
32
|
+
"max_inflight" => 0
|
|
33
|
+
}.freeze
|
|
34
|
+
|
|
35
|
+
# Lua script that raises fields to a new high-water mark. A field is only
|
|
36
|
+
# written when the new value is higher, so processes recording their own
|
|
37
|
+
# marks concurrently converge on the highest one.
|
|
38
|
+
#
|
|
39
|
+
# KEYS[1] = totals key
|
|
40
|
+
# ARGV = alternating field and value
|
|
41
|
+
RECORD_MAXIMA_SCRIPT = <<~LUA
|
|
42
|
+
for i = 1, #ARGV, 2 do
|
|
43
|
+
local field = ARGV[i]
|
|
44
|
+
local value = tonumber(ARGV[i + 1])
|
|
45
|
+
local current = redis.call('HGET', KEYS[1], field)
|
|
46
|
+
if not current or value > tonumber(current) then
|
|
47
|
+
redis.call('HSET', KEYS[1], field, value)
|
|
48
|
+
end
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
return 1
|
|
52
|
+
LUA
|
|
53
|
+
RECORD_MAXIMA_SHA = Digest::SHA1.hexdigest(RECORD_MAXIMA_SCRIPT).freeze
|
|
54
|
+
|
|
17
55
|
def initialize(config = nil)
|
|
18
56
|
@hostname = ::Socket.gethostname.force_encoding("UTF-8").freeze
|
|
19
57
|
@pid = ::Process.pid
|
|
20
58
|
@config = config
|
|
59
|
+
@mutex = Mutex.new
|
|
60
|
+
@pending = Hash.new(0)
|
|
61
|
+
@maxima = Hash.new(0)
|
|
62
|
+
@maxima_changed = false
|
|
63
|
+
@last_flush = monotonic_time
|
|
21
64
|
end
|
|
22
65
|
|
|
23
66
|
# Record a completed request.
|
|
24
67
|
#
|
|
25
68
|
# @param status [Integer, nil] HTTP response status code
|
|
26
69
|
# @param duration [Float] request duration in seconds
|
|
70
|
+
# @param processor_name [String, Symbol, nil] name of the processor that ran
|
|
71
|
+
# the request; when given, per-processor fields are recorded as well
|
|
27
72
|
# @return [void]
|
|
28
|
-
def record_request(status, duration)
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
73
|
+
def record_request(status, duration, processor_name: nil)
|
|
74
|
+
processor = processor_field_prefix(processor_name)
|
|
75
|
+
record do |pending|
|
|
76
|
+
pending["requests"] += 1
|
|
77
|
+
pending["duration"] += duration.to_f
|
|
78
|
+
pending["http_status:#{status}"] += 1 if status && status >= 100 && status < 600
|
|
79
|
+
if processor
|
|
80
|
+
pending["#{processor}requests"] += 1
|
|
81
|
+
pending["#{processor}duration"] += duration.to_f
|
|
35
82
|
end
|
|
36
83
|
end
|
|
37
|
-
rescue => e
|
|
38
|
-
handle_error(e)
|
|
39
84
|
end
|
|
40
85
|
|
|
41
86
|
# Record a request error
|
|
42
87
|
#
|
|
43
88
|
# @param error_type [String] the type of error that occurred
|
|
89
|
+
# @param processor_name [String, Symbol, nil] name of the processor that ran
|
|
90
|
+
# the request; when given, per-processor fields are recorded as well
|
|
44
91
|
# @return [void]
|
|
45
|
-
def record_error(error_type)
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
92
|
+
def record_error(error_type, processor_name: nil)
|
|
93
|
+
processor = processor_field_prefix(processor_name)
|
|
94
|
+
record do |pending|
|
|
95
|
+
pending["errors"] += 1
|
|
96
|
+
pending["errors:#{error_type}"] += 1
|
|
97
|
+
pending["#{processor}errors"] += 1 if processor
|
|
98
|
+
end
|
|
99
|
+
end
|
|
100
|
+
|
|
101
|
+
# Record the number of requests a processor had in flight, keeping the
|
|
102
|
+
# highest value seen. The count only rises when a request is handed to a
|
|
103
|
+
# processor, so recording it at that moment captures every high-water
|
|
104
|
+
# mark exactly.
|
|
105
|
+
#
|
|
106
|
+
# @param count [Integer] the number of requests in flight
|
|
107
|
+
# @param processor_name [String, Symbol, nil] name of the processor
|
|
108
|
+
# @return [void]
|
|
109
|
+
def record_inflight_peak(count, processor_name:)
|
|
110
|
+
processor = processor_field_prefix(processor_name)
|
|
111
|
+
return unless processor
|
|
112
|
+
|
|
113
|
+
field = "#{processor}max_inflight"
|
|
114
|
+
@mutex.synchronize do
|
|
115
|
+
next unless count > @maxima[field]
|
|
116
|
+
|
|
117
|
+
@maxima[field] = count
|
|
118
|
+
@maxima_changed = true
|
|
52
119
|
end
|
|
53
|
-
rescue => e
|
|
54
|
-
handle_error(e)
|
|
55
120
|
end
|
|
56
121
|
|
|
57
122
|
# Record a that a request was refused because the max capacity of the Processor was reached.
|
|
58
123
|
#
|
|
124
|
+
# @param processor_name [String, Symbol, nil] name of the processor that
|
|
125
|
+
# refused the request; when given, per-processor fields are recorded as well
|
|
59
126
|
# @return [void]
|
|
60
|
-
def record_capacity_exceeded
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
127
|
+
def record_capacity_exceeded(processor_name: nil)
|
|
128
|
+
processor = processor_field_prefix(processor_name)
|
|
129
|
+
record do |pending|
|
|
130
|
+
pending["max_capacity_exceeded"] += 1
|
|
131
|
+
pending["#{processor}max_capacity_exceeded"] += 1 if processor
|
|
132
|
+
end
|
|
133
|
+
end
|
|
134
|
+
|
|
135
|
+
# Flush pending deltas to Redis in a single pipelined write.
|
|
136
|
+
#
|
|
137
|
+
# @return [void]
|
|
138
|
+
def flush
|
|
139
|
+
pending = nil
|
|
140
|
+
maxima = nil
|
|
141
|
+
@mutex.synchronize do
|
|
142
|
+
@last_flush = monotonic_time
|
|
143
|
+
pending, @pending = @pending, Hash.new(0)
|
|
144
|
+
# The high-water marks are not consumed: they are the highest values
|
|
145
|
+
# this process has seen, and they are sent on every flush so that
|
|
146
|
+
# they are restored after the totals are cleared.
|
|
147
|
+
maxima = @maxima.dup
|
|
148
|
+
@maxima_changed = false
|
|
149
|
+
end
|
|
150
|
+
# A processor that is saturated records new marks without completing
|
|
151
|
+
# anything, so a new mark is a reason to flush on its own.
|
|
152
|
+
return if pending.empty? && maxima.empty?
|
|
153
|
+
|
|
154
|
+
begin
|
|
155
|
+
# The pipeline is a batch of increments, so it must not be replayed
|
|
156
|
+
# after a connection failure: the server may already have applied it
|
|
157
|
+
# and a replay would double count. A failure merges the deltas back
|
|
158
|
+
# instead, which at worst loses them if the write did land.
|
|
159
|
+
PatientHttp::Sidekiq.redis(retry_on_connection_error: false) do |redis|
|
|
160
|
+
redis.pipelined do |pipeline|
|
|
161
|
+
pending.each do |field, delta|
|
|
162
|
+
if float_field?(field)
|
|
163
|
+
pipeline.hincrbyfloat(TOTALS_KEY, field, delta)
|
|
164
|
+
else
|
|
165
|
+
pipeline.hincrby(TOTALS_KEY, field, delta)
|
|
166
|
+
end
|
|
167
|
+
end
|
|
168
|
+
pipeline.expire(TOTALS_KEY, TOTALS_TTL)
|
|
169
|
+
end
|
|
170
|
+
end
|
|
171
|
+
flush_maxima(maxima)
|
|
172
|
+
rescue => e
|
|
173
|
+
# Put the deltas back so nothing is lost; they are retried on the
|
|
174
|
+
# next flush.
|
|
175
|
+
@mutex.synchronize do
|
|
176
|
+
pending.each { |field, delta| @pending[field] += delta }
|
|
177
|
+
@maxima_changed = true if maxima.any?
|
|
65
178
|
end
|
|
179
|
+
handle_error(e)
|
|
180
|
+
end
|
|
181
|
+
end
|
|
182
|
+
|
|
183
|
+
# Flush when the configured interval has elapsed since the last flush.
|
|
184
|
+
# Called periodically by the task monitor thread.
|
|
185
|
+
#
|
|
186
|
+
# @return [void]
|
|
187
|
+
def flush_if_due
|
|
188
|
+
due = @mutex.synchronize do
|
|
189
|
+
(@pending.any? || @maxima_changed) && (monotonic_time - @last_flush >= flush_interval)
|
|
66
190
|
end
|
|
67
|
-
|
|
68
|
-
handle_error(e)
|
|
191
|
+
flush if due
|
|
69
192
|
end
|
|
70
193
|
|
|
71
194
|
# Get running totals
|
|
72
195
|
#
|
|
73
196
|
# @return [Hash] hash with requests, duration, errors, max_capacity_exceeded, http_status_counts
|
|
74
197
|
def get_totals
|
|
75
|
-
|
|
198
|
+
# Flush first so this process's own recorded events are visible.
|
|
199
|
+
# Other processes' unflushed deltas are stale by at most their flush
|
|
200
|
+
# interval.
|
|
201
|
+
flush
|
|
202
|
+
|
|
203
|
+
PatientHttp::Sidekiq.redis do |redis|
|
|
76
204
|
stats = redis.hgetall(TOTALS_KEY)
|
|
77
205
|
|
|
78
|
-
# Extract HTTP status counts
|
|
206
|
+
# Extract HTTP status counts, error type counts, and per-processor counts
|
|
79
207
|
http_status_counts = {}
|
|
80
208
|
error_type_counts = {}
|
|
209
|
+
processor_counts = {}
|
|
81
210
|
stats.each do |key, value|
|
|
82
211
|
if key.start_with?("http_status:")
|
|
83
212
|
status = key.sub("http_status:", "").to_i
|
|
@@ -85,10 +214,16 @@ module PatientHttp
|
|
|
85
214
|
elsif key.start_with?("errors:") && key != "errors"
|
|
86
215
|
error_type = key.sub("errors:", "")
|
|
87
216
|
error_type_counts[error_type] = value.to_i
|
|
217
|
+
elsif key.start_with?("processor:")
|
|
218
|
+
_, name, metric = key.split(":", 3)
|
|
219
|
+
next unless name && metric
|
|
220
|
+
|
|
221
|
+
counts = (processor_counts[name] ||= PROCESSOR_METRICS.dup)
|
|
222
|
+
counts[metric] = (metric == "duration") ? value.to_f.round(6) : value.to_i
|
|
88
223
|
end
|
|
89
224
|
end
|
|
90
225
|
|
|
91
|
-
{
|
|
226
|
+
totals = {
|
|
92
227
|
"requests" => (stats["requests"] || 0).to_i,
|
|
93
228
|
"duration" => (stats["duration"] || 0).to_f.round(6),
|
|
94
229
|
"errors" => (stats["errors"] || 0).to_i,
|
|
@@ -96,6 +231,8 @@ module PatientHttp
|
|
|
96
231
|
"http_status_counts" => http_status_counts.sort.to_h,
|
|
97
232
|
"error_type_counts" => error_type_counts.sort.to_h
|
|
98
233
|
}
|
|
234
|
+
totals["processors"] = processor_counts.sort.to_h if processor_counts.any?
|
|
235
|
+
totals
|
|
99
236
|
end
|
|
100
237
|
end
|
|
101
238
|
|
|
@@ -103,13 +240,73 @@ module PatientHttp
|
|
|
103
240
|
#
|
|
104
241
|
# @return [void]
|
|
105
242
|
def reset!
|
|
106
|
-
|
|
243
|
+
@mutex.synchronize do
|
|
244
|
+
@pending = Hash.new(0)
|
|
245
|
+
@maxima = Hash.new(0)
|
|
246
|
+
@maxima_changed = false
|
|
247
|
+
@last_flush = monotonic_time
|
|
248
|
+
end
|
|
249
|
+
PatientHttp::Sidekiq.redis do |redis|
|
|
107
250
|
redis.del(TOTALS_KEY)
|
|
108
251
|
end
|
|
109
252
|
end
|
|
110
253
|
|
|
111
254
|
private
|
|
112
255
|
|
|
256
|
+
# Apply increments under the mutex; flush synchronously when the flush
|
|
257
|
+
# interval is 0 (the compatibility mode where every event writes through
|
|
258
|
+
# to Redis immediately).
|
|
259
|
+
def record
|
|
260
|
+
@mutex.synchronize do
|
|
261
|
+
yield @pending
|
|
262
|
+
end
|
|
263
|
+
flush if flush_interval.zero?
|
|
264
|
+
end
|
|
265
|
+
|
|
266
|
+
# Raise the stored high-water marks to this process's values. A separate
|
|
267
|
+
# call because a maximum cannot be pipelined with the increments: it is a
|
|
268
|
+
# read and a conditional write, which has to happen inside the server.
|
|
269
|
+
#
|
|
270
|
+
# @param maxima [Hash] field to value
|
|
271
|
+
# @return [void]
|
|
272
|
+
def flush_maxima(maxima)
|
|
273
|
+
return if maxima.empty?
|
|
274
|
+
|
|
275
|
+
argv = maxima.flat_map { |field, value| [field, value.to_s] }
|
|
276
|
+
PatientHttp::Sidekiq.redis(retry_on_connection_error: false) do |redis|
|
|
277
|
+
redis.call("EVALSHA", RECORD_MAXIMA_SHA, 1, TOTALS_KEY, *argv)
|
|
278
|
+
rescue RedisClient::CommandError => e
|
|
279
|
+
raise unless e.message.include?("NOSCRIPT")
|
|
280
|
+
|
|
281
|
+
redis.call("EVAL", RECORD_MAXIMA_SCRIPT, 1, TOTALS_KEY, *argv)
|
|
282
|
+
end
|
|
283
|
+
end
|
|
284
|
+
|
|
285
|
+
def flush_interval
|
|
286
|
+
@config&.stats_flush_interval || 5
|
|
287
|
+
end
|
|
288
|
+
|
|
289
|
+
# Field name prefix for a processor's own counters, or nil when the
|
|
290
|
+
# counters would carry no information. A single processor profile
|
|
291
|
+
# duplicates the overall totals, so its fields are left out to keep the
|
|
292
|
+
# totals hash small.
|
|
293
|
+
#
|
|
294
|
+
# Colons separate the field name components, so they are removed from
|
|
295
|
+
# the processor name.
|
|
296
|
+
#
|
|
297
|
+
# @param processor_name [String, Symbol, nil] the processor name
|
|
298
|
+
# @return [String, nil] the prefix, or nil to skip per-processor fields
|
|
299
|
+
def processor_field_prefix(processor_name)
|
|
300
|
+
return nil if processor_name.nil?
|
|
301
|
+
return nil if @config && !@config.multiple_processors?
|
|
302
|
+
|
|
303
|
+
"processor:#{processor_name.to_s.tr(":", "-")}:"
|
|
304
|
+
end
|
|
305
|
+
|
|
306
|
+
def float_field?(field)
|
|
307
|
+
field == "duration" || field.end_with?(":duration")
|
|
308
|
+
end
|
|
309
|
+
|
|
113
310
|
def handle_error(error)
|
|
114
311
|
@config&.logger&.error("[PatientHttp::Sidekiq] Stats error: #{error.inspect}")
|
|
115
312
|
raise error if PatientHttp.testing?
|
|
@@ -28,7 +28,9 @@ module PatientHttp
|
|
|
28
28
|
# @return [void]
|
|
29
29
|
def on_complete(response, callback)
|
|
30
30
|
data = store_if_needed(response.as_json)
|
|
31
|
-
|
|
31
|
+
PatientHttp::Sidekiq.with_redis_pool do
|
|
32
|
+
callback_worker.perform_async(data, "response", callback)
|
|
33
|
+
end
|
|
32
34
|
delete_stored_request_payload
|
|
33
35
|
end
|
|
34
36
|
|
|
@@ -42,7 +44,9 @@ module PatientHttp
|
|
|
42
44
|
# @return [void]
|
|
43
45
|
def on_error(error, callback)
|
|
44
46
|
data = store_if_needed(error.as_json)
|
|
45
|
-
|
|
47
|
+
PatientHttp::Sidekiq.with_redis_pool do
|
|
48
|
+
callback_worker.perform_async(data, "error", callback)
|
|
49
|
+
end
|
|
46
50
|
delete_stored_request_payload
|
|
47
51
|
end
|
|
48
52
|
|
|
@@ -50,7 +54,9 @@ module PatientHttp
|
|
|
50
54
|
#
|
|
51
55
|
# @return [String] the job ID
|
|
52
56
|
def retry
|
|
53
|
-
::Sidekiq
|
|
57
|
+
PatientHttp::Sidekiq.with_redis_pool do
|
|
58
|
+
::Sidekiq::Client.push(@sidekiq_job)
|
|
59
|
+
end
|
|
54
60
|
end
|
|
55
61
|
|
|
56
62
|
# Return the job ID from the Sidekiq job.
|