rails_error_dashboard 0.12.1 → 0.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/app/controllers/rails_error_dashboard/application_controller.rb +92 -11
- data/app/controllers/rails_error_dashboard/errors_controller.rb +111 -34
- data/app/controllers/rails_error_dashboard/webhooks_controller.rb +3 -0
- data/app/helpers/rails_error_dashboard/application_helper.rb +46 -1
- data/app/helpers/rails_error_dashboard/backtrace_helper.rb +8 -0
- data/app/jobs/rails_error_dashboard/concerns/plain_channel_message.rb +71 -0
- data/app/jobs/rails_error_dashboard/notification_burst_summary_job.rb +55 -0
- data/app/jobs/rails_error_dashboard/retention_cleanup_job.rb +68 -1
- data/app/jobs/rails_error_dashboard/storm_notification_job.rb +8 -44
- data/app/models/rails_error_dashboard/error_baseline.rb +12 -9
- data/app/models/rails_error_dashboard/error_comment.rb +0 -5
- data/app/models/rails_error_dashboard/error_log.rb +9 -3
- data/app/models/rails_error_dashboard/error_logs_record.rb +34 -0
- data/app/models/rails_error_dashboard/error_occurrence.rb +9 -1
- data/app/views/layouts/rails_error_dashboard.html.erb +11 -3
- data/app/views/rails_error_dashboard/errors/_discussion.html.erb +18 -11
- data/app/views/rails_error_dashboard/errors/_issue_section.html.erb +22 -8
- data/app/views/rails_error_dashboard/errors/_sidebar_metadata.html.erb +15 -6
- data/app/views/rails_error_dashboard/errors/analytics.html.erb +8 -8
- data/app/views/rails_error_dashboard/errors/diagnostic_dumps.html.erb +1 -1
- data/app/views/rails_error_dashboard/errors/index.html.erb +6 -1
- data/app/views/rails_error_dashboard/errors/platform_comparison.html.erb +7 -7
- data/app/views/rails_error_dashboard/errors/releases.html.erb +2 -2
- data/app/views/rails_error_dashboard/errors/settings.html.erb +2 -0
- data/config/locales/de.yml +28 -0
- data/config/locales/en.yml +35 -0
- data/config/locales/es.yml +28 -0
- data/config/locales/fr.yml +28 -0
- data/config/locales/it.yml +28 -0
- data/config/locales/ja.yml +28 -0
- data/config/locales/pl.yml +28 -0
- data/config/locales/pt-BR.yml +28 -0
- data/config/locales/ru.yml +28 -0
- data/config/locales/uk.yml +28 -0
- data/config/locales/zh-CN.yml +28 -0
- data/db/migrate/20260917000001_add_last_notified_at_to_error_logs.rb +40 -0
- data/lib/generators/rails_error_dashboard/install/templates/initializer.rb +12 -1
- data/lib/rails_error_dashboard/commands/assign_error.rb +12 -2
- data/lib/rails_error_dashboard/commands/backfill_environments.rb +2 -0
- data/lib/rails_error_dashboard/commands/backfill_resolved_at.rb +42 -0
- data/lib/rails_error_dashboard/commands/batch_delete_errors.rb +1 -0
- data/lib/rails_error_dashboard/commands/batch_mute_errors.rb +2 -0
- data/lib/rails_error_dashboard/commands/batch_resolve_errors.rb +2 -0
- data/lib/rails_error_dashboard/commands/batch_unmute_errors.rb +2 -0
- data/lib/rails_error_dashboard/commands/find_or_increment_error.rb +39 -7
- data/lib/rails_error_dashboard/commands/flush_rack_attack_events.rb +3 -1
- data/lib/rails_error_dashboard/commands/flush_storm_counts.rb +27 -4
- data/lib/rails_error_dashboard/commands/flush_swallowed_exceptions.rb +5 -2
- data/lib/rails_error_dashboard/commands/link_existing_issue.rb +1 -0
- data/lib/rails_error_dashboard/commands/log_error.rb +82 -23
- data/lib/rails_error_dashboard/commands/mute_error.rb +1 -0
- data/lib/rails_error_dashboard/commands/resolve_error.rb +2 -0
- data/lib/rails_error_dashboard/commands/scrub_invalid_encoding.rb +104 -0
- data/lib/rails_error_dashboard/commands/snooze_error.rb +34 -10
- data/lib/rails_error_dashboard/commands/unmute_error.rb +1 -0
- data/lib/rails_error_dashboard/commands/update_error_priority.rb +23 -2
- data/lib/rails_error_dashboard/commands/update_error_status.rb +33 -6
- data/lib/rails_error_dashboard/configuration.rb +19 -1
- data/lib/rails_error_dashboard/engine.rb +15 -0
- data/lib/rails_error_dashboard/queries/analytics_stats.rb +4 -3
- data/lib/rails_error_dashboard/queries/baseline_stats.rb +107 -0
- data/lib/rails_error_dashboard/queries/dashboard_stats.rb +55 -50
- data/lib/rails_error_dashboard/queries/error_correlation.rb +6 -3
- data/lib/rails_error_dashboard/queries/errors_list.rb +15 -2
- data/lib/rails_error_dashboard/queries/similar_errors.rb +1 -1
- data/lib/rails_error_dashboard/services/analytics_cache_manager.rb +43 -18
- data/lib/rails_error_dashboard/services/backtrace_processor.rb +3 -1
- data/lib/rails_error_dashboard/services/cause_chain_extractor.rb +3 -1
- data/lib/rails_error_dashboard/services/codeberg_issue_client.rb +13 -2
- data/lib/rails_error_dashboard/services/diagnostic_dump_generator.rb +5 -3
- data/lib/rails_error_dashboard/services/encoding_sanitizer.rb +80 -0
- data/lib/rails_error_dashboard/services/error_broadcaster.rb +180 -32
- data/lib/rails_error_dashboard/services/error_hash_generator.rb +9 -3
- data/lib/rails_error_dashboard/services/error_notification_dispatcher.rb +13 -0
- data/lib/rails_error_dashboard/services/exception_filter.rb +55 -0
- data/lib/rails_error_dashboard/services/git_head_reader.rb +102 -0
- data/lib/rails_error_dashboard/services/notification_throttler.rb +172 -28
- data/lib/rails_error_dashboard/services/sensitive_data_filter.rb +47 -1
- data/lib/rails_error_dashboard/services/storm_protection/circuit_breaker.rb +59 -3
- data/lib/rails_error_dashboard/services/storm_protection/fingerprint_buckets.rb +25 -2
- data/lib/rails_error_dashboard/services/storm_protection/gate.rb +32 -5
- data/lib/rails_error_dashboard/services/swallowed_exception_tracker.rb +99 -23
- data/lib/rails_error_dashboard/services/url_safety.rb +40 -0
- data/lib/rails_error_dashboard/subscribers/issue_tracker_subscriber.rb +11 -2
- data/lib/rails_error_dashboard/value_objects/error_context.rb +3 -1
- data/lib/rails_error_dashboard/version.rb +1 -1
- data/lib/rails_error_dashboard.rb +31 -0
- data/lib/tasks/error_dashboard.rake +54 -4
- metadata +10 -2
|
@@ -2,21 +2,73 @@
|
|
|
2
2
|
|
|
3
3
|
module RailsErrorDashboard
|
|
4
4
|
module Services
|
|
5
|
-
#
|
|
5
|
+
# Throttle error notifications to prevent alert fatigue
|
|
6
6
|
#
|
|
7
7
|
# Checks severity minimum, per-error cooldown, and threshold milestones.
|
|
8
|
-
#
|
|
8
|
+
#
|
|
9
|
+
# The cooldown is CLAIMED IN THE DATABASE (error_logs.last_notified_at, see
|
|
10
|
+
# claim!). A per-process Hash cannot do this job: every Puma worker and
|
|
11
|
+
# every job process has its own, so one error reopened by a bad deploy
|
|
12
|
+
# notified once per process, and a restart forgot the cooldown altogether.
|
|
13
|
+
# The Hash survives only as a bounded fallback for the window between
|
|
14
|
+
# upgrading the gem and running its migration, and for callers that pass
|
|
15
|
+
# something other than a persisted row.
|
|
16
|
+
#
|
|
9
17
|
# Thread-safe via Mutex. Fail-open: returns true on any error.
|
|
10
18
|
class NotificationThrottler
|
|
11
19
|
# Severity levels ranked from lowest to highest
|
|
12
20
|
SEVERITY_RANK = { low: 0, medium: 1, high: 2, critical: 3 }.freeze
|
|
13
21
|
|
|
22
|
+
# Hard cap on the in-process fallback. Expired entries are swept on every
|
|
23
|
+
# insert; the cap only bites when more than this many distinct errors are
|
|
24
|
+
# inside their cooldown at once, and then the oldest goes first.
|
|
25
|
+
MAX_TRACKED = 1_000
|
|
26
|
+
|
|
27
|
+
COOLDOWN_COLUMN = "last_notified_at"
|
|
28
|
+
|
|
14
29
|
@last_notification_times = {}
|
|
15
30
|
@mutex = Mutex.new
|
|
31
|
+
@burst_mutex = Mutex.new
|
|
32
|
+
@burst_window_start = nil
|
|
33
|
+
@burst_count = 0
|
|
16
34
|
|
|
17
35
|
class << self
|
|
18
|
-
#
|
|
19
|
-
#
|
|
36
|
+
# Take the right to notify about this error, atomically.
|
|
37
|
+
#
|
|
38
|
+
# Call it immediately BEFORE dispatching, and dispatch only on true. It
|
|
39
|
+
# is claim-then-send: a claim followed by a failed send is not handed
|
|
40
|
+
# back, so that error stays quiet until the cooldown ends. The
|
|
41
|
+
# alternative (send, then record) is what let N processes all send.
|
|
42
|
+
#
|
|
43
|
+
# With the column present this is ONE statement and no read:
|
|
44
|
+
#
|
|
45
|
+
# UPDATE error_logs SET last_notified_at = :now
|
|
46
|
+
# WHERE id = :id AND (last_notified_at IS NULL OR last_notified_at < :cutoff)
|
|
47
|
+
#
|
|
48
|
+
# Exactly one connection can match the row, on SQLite, PostgreSQL and
|
|
49
|
+
# MySQL alike, so exactly one process notifies per cooldown window.
|
|
50
|
+
#
|
|
51
|
+
# @param error_log [ErrorLog] the row being notified about
|
|
52
|
+
# @param respect_cooldown [Boolean] false for notifications the cooldown
|
|
53
|
+
# has never applied to (first occurrence, threshold milestones): always
|
|
54
|
+
# granted, but still stamped, so a reopen soon afterwards is throttled
|
|
55
|
+
# @return [Boolean] true if the caller may notify
|
|
56
|
+
def claim!(error_log, respect_cooldown: true)
|
|
57
|
+
minutes = respect_cooldown ? cooldown_minutes : 0
|
|
58
|
+
|
|
59
|
+
if database_claim?(error_log)
|
|
60
|
+
claim_in_database(error_log, minutes)
|
|
61
|
+
else
|
|
62
|
+
claim_in_process(error_log, minutes)
|
|
63
|
+
end
|
|
64
|
+
rescue => e
|
|
65
|
+
# Fail-open: a throttler that cannot decide must not lose a page.
|
|
66
|
+
RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] NotificationThrottler.claim! failed: #{e.class}: #{e.message}")
|
|
67
|
+
true
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
# Should we send a notification for this error? Read-only (severity
|
|
71
|
+
# minimum + cooldown): it takes nothing. claim! is what notifies.
|
|
20
72
|
# @param error_log [ErrorLog] The error to check
|
|
21
73
|
# @return [Boolean] true if notification should be sent
|
|
22
74
|
def should_notify?(error_log)
|
|
@@ -79,53 +131,145 @@ module RailsErrorDashboard
|
|
|
79
131
|
false
|
|
80
132
|
end
|
|
81
133
|
|
|
82
|
-
#
|
|
83
|
-
#
|
|
84
|
-
|
|
85
|
-
|
|
134
|
+
# May a FIRST-OCCURRENCE notification go out right now?
|
|
135
|
+
#
|
|
136
|
+
# The per-error cooldown cannot bound a bad deploy that produces hundreds
|
|
137
|
+
# of DISTINCT new errors: each is a first occurrence, and first
|
|
138
|
+
# occurrences always notify. This is a fixed window counter:
|
|
139
|
+
#
|
|
140
|
+
# :notify within config.notification_burst_limit for this window
|
|
141
|
+
# :summarize the first one over the limit: send ONE summary instead
|
|
142
|
+
# :suppress everything after that, until the window ends
|
|
143
|
+
#
|
|
144
|
+
# Per process, like Gate.issue_creation_allowed?, because there is no
|
|
145
|
+
# store every deployment shares except the database and this is asked on
|
|
146
|
+
# the capture path. Worst case is limit x processes per window.
|
|
147
|
+
#
|
|
148
|
+
# Each call consumes a slot: ask only when about to notify. A limit or
|
|
149
|
+
# window of 0 / nil turns the cap off. Fails open to :notify.
|
|
150
|
+
#
|
|
151
|
+
# @return [Symbol] :notify, :summarize or :suppress
|
|
152
|
+
def burst_decision
|
|
153
|
+
limit = RailsErrorDashboard.configuration.notification_burst_limit.to_i
|
|
154
|
+
window = RailsErrorDashboard.configuration.notification_burst_window_seconds.to_i
|
|
155
|
+
return :notify unless limit.positive? && window.positive?
|
|
86
156
|
|
|
87
|
-
|
|
88
|
-
|
|
157
|
+
now = monotonic_now
|
|
158
|
+
count = @burst_mutex.synchronize do
|
|
159
|
+
if @burst_window_start.nil? || now - @burst_window_start >= window
|
|
160
|
+
@burst_window_start = now
|
|
161
|
+
@burst_count = 0
|
|
162
|
+
end
|
|
163
|
+
@burst_count += 1
|
|
164
|
+
end
|
|
165
|
+
|
|
166
|
+
if count <= limit
|
|
167
|
+
:notify
|
|
168
|
+
elsif count == limit + 1
|
|
169
|
+
:summarize
|
|
170
|
+
else
|
|
171
|
+
:suppress
|
|
89
172
|
end
|
|
90
173
|
rescue => e
|
|
91
|
-
RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] NotificationThrottler.
|
|
174
|
+
RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] NotificationThrottler.burst_decision failed: #{e.class}: #{e.message}")
|
|
175
|
+
:notify
|
|
176
|
+
end
|
|
177
|
+
|
|
178
|
+
# Record that a notification was sent for this error, unconditionally.
|
|
179
|
+
# Kept for callers that notify outside LogError; LogError itself uses
|
|
180
|
+
# claim!, which decides and records in one step.
|
|
181
|
+
# @param error_log [ErrorLog] The error that was notified about
|
|
182
|
+
def record_notification(error_log)
|
|
183
|
+
claim!(error_log, respect_cooldown: false)
|
|
184
|
+
nil
|
|
92
185
|
end
|
|
93
186
|
|
|
94
|
-
# Clear all throttle state (for testing)
|
|
187
|
+
# Clear all in-process throttle state (for testing)
|
|
95
188
|
def clear!
|
|
96
189
|
@mutex.synchronize do
|
|
97
190
|
@last_notification_times.clear
|
|
98
191
|
end
|
|
192
|
+
@burst_mutex.synchronize do
|
|
193
|
+
@burst_window_start = nil
|
|
194
|
+
@burst_count = 0
|
|
195
|
+
end
|
|
99
196
|
end
|
|
100
197
|
|
|
101
|
-
|
|
102
|
-
# @param max_age_hours [Integer] Remove entries older than this (default: 24)
|
|
103
|
-
def cleanup!(max_age_hours: 24)
|
|
104
|
-
cutoff_time = max_age_hours.hours.ago
|
|
198
|
+
private
|
|
105
199
|
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
end
|
|
200
|
+
def monotonic_now
|
|
201
|
+
Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
109
202
|
end
|
|
110
203
|
|
|
111
|
-
|
|
204
|
+
def cooldown_minutes
|
|
205
|
+
RailsErrorDashboard.configuration.notification_cooldown_minutes.to_i
|
|
206
|
+
end
|
|
112
207
|
|
|
113
|
-
#
|
|
114
|
-
#
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
208
|
+
# A persisted row AND the column: before the migration has run the
|
|
209
|
+
# UPDATE would raise on every notification.
|
|
210
|
+
def database_claim?(error_log)
|
|
211
|
+
error_log.is_a?(ErrorLog) && error_log.persisted? &&
|
|
212
|
+
ErrorLog.column_names.include?(COOLDOWN_COLUMN)
|
|
213
|
+
end
|
|
119
214
|
|
|
215
|
+
def claim_in_database(error_log, minutes)
|
|
216
|
+
now = Time.current
|
|
217
|
+
scope = ErrorLog.where(id: error_log.id)
|
|
218
|
+
if minutes.positive?
|
|
219
|
+
scope = scope.where("#{COOLDOWN_COLUMN} IS NULL OR #{COOLDOWN_COLUMN} < ?", now - minutes.minutes)
|
|
220
|
+
end
|
|
221
|
+
|
|
222
|
+
granted = scope.update_all(COOLDOWN_COLUMN => now) == 1
|
|
223
|
+
# No cooldown to lose: a row deleted under us must not silence the page.
|
|
224
|
+
granted || !minutes.positive?
|
|
225
|
+
end
|
|
226
|
+
|
|
227
|
+
def claim_in_process(error_log, minutes)
|
|
120
228
|
key = error_log.error_hash
|
|
229
|
+
now = Time.current
|
|
121
230
|
|
|
122
231
|
@mutex.synchronize do
|
|
123
232
|
last_time = @last_notification_times[key]
|
|
124
|
-
|
|
233
|
+
next false if minutes.positive? && last_time && now <= last_time + minutes.minutes
|
|
125
234
|
|
|
126
|
-
|
|
235
|
+
remember(key, now)
|
|
236
|
+
true
|
|
127
237
|
end
|
|
128
238
|
end
|
|
239
|
+
|
|
240
|
+
# Caller holds @mutex. Delete-and-reinsert keeps the Hash in recency
|
|
241
|
+
# order, so "oldest" is simply the first key.
|
|
242
|
+
def remember(key, now)
|
|
243
|
+
window = cooldown_minutes
|
|
244
|
+
@last_notification_times.delete(key)
|
|
245
|
+
# With no cooldown nothing will ever read the entry back.
|
|
246
|
+
return unless window.positive?
|
|
247
|
+
|
|
248
|
+
cutoff = now - window.minutes
|
|
249
|
+
@last_notification_times.delete_if { |_, time| time < cutoff }
|
|
250
|
+
@last_notification_times.shift while @last_notification_times.size >= MAX_TRACKED
|
|
251
|
+
@last_notification_times[key] = now
|
|
252
|
+
end
|
|
253
|
+
|
|
254
|
+
# Is the error outside the cooldown window? Read-only.
|
|
255
|
+
# @param error_log [ErrorLog] The error to check
|
|
256
|
+
# @return [Boolean] true if not in cooldown (ok to notify)
|
|
257
|
+
def cooldown_ok?(error_log)
|
|
258
|
+
minutes = cooldown_minutes
|
|
259
|
+
return true unless minutes.positive?
|
|
260
|
+
|
|
261
|
+
last_time =
|
|
262
|
+
if database_claim?(error_log)
|
|
263
|
+
# Read the ROW, not the object: claim! stamps it with update_all,
|
|
264
|
+
# so the caller's copy -- and every other process's -- is stale.
|
|
265
|
+
ErrorLog.where(id: error_log.id).pick(COOLDOWN_COLUMN)
|
|
266
|
+
else
|
|
267
|
+
@mutex.synchronize { @last_notification_times[error_log.error_hash] }
|
|
268
|
+
end
|
|
269
|
+
return true if last_time.nil?
|
|
270
|
+
|
|
271
|
+
Time.current > (last_time + minutes.minutes)
|
|
272
|
+
end
|
|
129
273
|
end
|
|
130
274
|
end
|
|
131
275
|
end
|
|
@@ -51,7 +51,53 @@ module RailsErrorDashboard
|
|
|
51
51
|
attributes
|
|
52
52
|
end
|
|
53
53
|
|
|
54
|
-
|
|
54
|
+
SESSION_DIGEST_PREFIX = "h1:"
|
|
55
|
+
SESSION_DIGEST_FORMAT = /\Ah1:\h{32}\z/
|
|
56
|
+
|
|
57
|
+
# Keyed digest of a session ID, for storage.
|
|
58
|
+
#
|
|
59
|
+
# A session ID is a bearer credential, and the only thing the dashboard does
|
|
60
|
+
# with it is equality ("these occurrences were one session"). An HMAC keeps
|
|
61
|
+
# that and is useless for replay; keying it with secret_key_base means a
|
|
62
|
+
# guessed ID cannot be confirmed offline either. The prefix tells a digest
|
|
63
|
+
# from a raw Rails session ID (itself 32 hex characters), which makes
|
|
64
|
+
# digesting idempotent and leaves room for an "h2:".
|
|
65
|
+
#
|
|
66
|
+
# Rotating secret_key_base breaks correlation across the rotation.
|
|
67
|
+
#
|
|
68
|
+
# @param raw [#to_s, nil] raw session ID, or an existing digest
|
|
69
|
+
# @return [String, nil] "h1:" + 32 hex characters; nil for blank input or on
|
|
70
|
+
# any failure (never the raw value, never an exception)
|
|
71
|
+
def self.digest_session_id(raw)
|
|
72
|
+
value = raw.to_s
|
|
73
|
+
return nil if value.empty?
|
|
74
|
+
# Exact shape only: a raw value that merely starts with "h1:" is digested.
|
|
75
|
+
return value if value.match?(SESSION_DIGEST_FORMAT)
|
|
76
|
+
|
|
77
|
+
SESSION_DIGEST_PREFIX + OpenSSL::HMAC.hexdigest("SHA256", session_digest_key, value)[0, 32]
|
|
78
|
+
rescue => e
|
|
79
|
+
RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] Session ID digest failed: #{e.class}")
|
|
80
|
+
nil
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
# What to write to error_occurrences.session_id: the digest while filtering
|
|
84
|
+
# is on, the raw value when the operator has turned filtering off.
|
|
85
|
+
def self.storable_session_id(raw)
|
|
86
|
+
return digest_session_id(raw) if RailsErrorDashboard.configuration.filter_sensitive_data
|
|
87
|
+
|
|
88
|
+
raw.nil? ? nil : raw.to_s
|
|
89
|
+
rescue => e
|
|
90
|
+
RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] Session ID handling failed: #{e.class}")
|
|
91
|
+
nil
|
|
92
|
+
end
|
|
93
|
+
|
|
94
|
+
def self.session_digest_key
|
|
95
|
+
key = Rails.application.secret_key_base if defined?(Rails) && Rails.application.respond_to?(:secret_key_base)
|
|
96
|
+
key.to_s.empty? ? "rails_error_dashboard" : key.to_s
|
|
97
|
+
rescue StandardError
|
|
98
|
+
"rails_error_dashboard"
|
|
99
|
+
end
|
|
100
|
+
|
|
55
101
|
# @return [ActiveSupport::ParameterFilter, nil]
|
|
56
102
|
def self.parameter_filter
|
|
57
103
|
@parameter_filter ||= build_parameter_filter
|
|
@@ -22,8 +22,15 @@ module RailsErrorDashboard
|
|
|
22
22
|
class CircuitBreaker
|
|
23
23
|
BUCKET_SECONDS = 10
|
|
24
24
|
CALM_BUCKETS_TO_CLOSE = 2
|
|
25
|
+
# Empty buckets replayed after a silence. Seven is enough to walk
|
|
26
|
+
# :open -> :half_open -> :closed; the cap keeps a roll O(1) however long
|
|
27
|
+
# the process sat idle.
|
|
28
|
+
MAX_CATCH_UP_BUCKETS = 7
|
|
25
29
|
|
|
26
|
-
|
|
30
|
+
# Incremented every time the breaker ENTERS :half_open. The gate compares
|
|
31
|
+
# it with the last value it saw to restart its probe counter, so each
|
|
32
|
+
# recovery attempt probes with its first event.
|
|
33
|
+
attr_reader :half_open_epoch
|
|
27
34
|
|
|
28
35
|
# @param clock [#call] returns monotonic seconds; injectable for tests
|
|
29
36
|
def initialize(clock: -> { Process.clock_gettime(Process::CLOCK_MONOTONIC) })
|
|
@@ -40,6 +47,7 @@ module RailsErrorDashboard
|
|
|
40
47
|
@calm_buckets = 0
|
|
41
48
|
@opened_at = nil
|
|
42
49
|
@episode = nil
|
|
50
|
+
@half_open_epoch = 0
|
|
43
51
|
end
|
|
44
52
|
end
|
|
45
53
|
|
|
@@ -61,6 +69,25 @@ module RailsErrorDashboard
|
|
|
61
69
|
@state
|
|
62
70
|
end
|
|
63
71
|
|
|
72
|
+
# Current state, advanced by elapsed TIME as well as by events. Without the
|
|
73
|
+
# tick a storm that simply stopped left the breaker open for ever: only
|
|
74
|
+
# record! rolled buckets, and nothing calls record! when errors stop.
|
|
75
|
+
def state
|
|
76
|
+
tick!
|
|
77
|
+
@state
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
# Roll the bucket if one is due. Cheap when it is not: one clock read and
|
|
81
|
+
# a comparison, no lock. Safe to call from anywhere, any number of times.
|
|
82
|
+
def tick!
|
|
83
|
+
now = @clock.call
|
|
84
|
+
roll!(now) if now - @bucket_start >= BUCKET_SECONDS
|
|
85
|
+
nil
|
|
86
|
+
rescue => e
|
|
87
|
+
RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] CircuitBreaker#tick! failed: #{e.class}: #{e.message}")
|
|
88
|
+
nil
|
|
89
|
+
end
|
|
90
|
+
|
|
64
91
|
# Episode metadata for the honesty layer (storm_events row).
|
|
65
92
|
# @return [Hash, nil] nil when no episode is active or recently closed
|
|
66
93
|
def episode_snapshot
|
|
@@ -81,10 +108,38 @@ module RailsErrorDashboard
|
|
|
81
108
|
elapsed = now - @bucket_start
|
|
82
109
|
return if elapsed < BUCKET_SECONDS # another thread already rolled
|
|
83
110
|
|
|
84
|
-
|
|
111
|
+
count = @bucket_count.value
|
|
112
|
+
bucket_start = @bucket_start
|
|
85
113
|
@bucket_start = now
|
|
86
114
|
@bucket_count = Concurrent::AtomicFixnum.new(0)
|
|
87
|
-
|
|
115
|
+
|
|
116
|
+
buckets = (elapsed / BUCKET_SECONDS).floor
|
|
117
|
+
if buckets <= 1
|
|
118
|
+
transition!(count / elapsed.to_f, now)
|
|
119
|
+
else
|
|
120
|
+
catch_up!(count, elapsed, bucket_start, now, buckets)
|
|
121
|
+
end
|
|
122
|
+
end
|
|
123
|
+
end
|
|
124
|
+
|
|
125
|
+
# More than one bucket of wall time has passed since the last roll, so
|
|
126
|
+
# nobody called in between: replay the gap instead of treating it as one
|
|
127
|
+
# long bucket. Each step carries the time its bucket ENDED, which is what
|
|
128
|
+
# keeps the cooldown and the two-calm-bucket rule honest (10s past the
|
|
129
|
+
# cooldown is :half_open, not :closed).
|
|
130
|
+
#
|
|
131
|
+
# The measured bucket keeps the diluted rate (count / elapsed) it has
|
|
132
|
+
# always had, so a burst followed by silence never escalates a closed
|
|
133
|
+
# breaker after the fact. Only the most recent MAX_CATCH_UP_BUCKETS empty
|
|
134
|
+
# buckets are replayed; older ones could not change the outcome.
|
|
135
|
+
def catch_up!(count, elapsed, bucket_start, now, buckets)
|
|
136
|
+
transition!(count / elapsed.to_f, bucket_start + BUCKET_SECONDS)
|
|
137
|
+
|
|
138
|
+
empty = [ buckets - 1, MAX_CATCH_UP_BUCKETS ].min
|
|
139
|
+
(empty - 1).downto(0) do |back|
|
|
140
|
+
break if @state == :closed
|
|
141
|
+
|
|
142
|
+
transition!(0.0, now - (back * BUCKET_SECONDS))
|
|
88
143
|
end
|
|
89
144
|
end
|
|
90
145
|
|
|
@@ -111,6 +166,7 @@ module RailsErrorDashboard
|
|
|
111
166
|
if now - @opened_at >= cooldown_seconds && rate < shedding_threshold
|
|
112
167
|
@state = :half_open
|
|
113
168
|
@calm_buckets = 0
|
|
169
|
+
@half_open_epoch += 1
|
|
114
170
|
end
|
|
115
171
|
when :half_open
|
|
116
172
|
if rate >= shedding_threshold
|
|
@@ -39,6 +39,7 @@ module RailsErrorDashboard
|
|
|
39
39
|
|
|
40
40
|
def reset!
|
|
41
41
|
@entries = Concurrent::Map.new
|
|
42
|
+
@last_sweep = nil
|
|
42
43
|
end
|
|
43
44
|
|
|
44
45
|
# Decide capture fidelity for one event of this fingerprint.
|
|
@@ -68,12 +69,34 @@ module RailsErrorDashboard
|
|
|
68
69
|
return existing if existing
|
|
69
70
|
|
|
70
71
|
# Bounded: never insert past the cap (size check is approximate
|
|
71
|
-
# under concurrency — a few entries over the cap is fine)
|
|
72
|
-
|
|
72
|
+
# under concurrency — a few entries over the cap is fine). Before
|
|
73
|
+
# declining, make room by dropping fingerprints that have gone quiet:
|
|
74
|
+
# without that the map filled once and stayed full for the life of
|
|
75
|
+
# the process, so the first N fingerprints a worker ever saw were the
|
|
76
|
+
# only ones it would ever rate-limit. If every tracked entry is still
|
|
77
|
+
# live, a new key is still declined — a storm of unique fingerprints
|
|
78
|
+
# is the global breaker's job (Layer 2), not this map's.
|
|
79
|
+
if @entries.size >= max_tracked
|
|
80
|
+
sweep_expired!(now)
|
|
81
|
+
return nil if @entries.size >= max_tracked
|
|
82
|
+
end
|
|
73
83
|
|
|
74
84
|
@entries.compute_if_absent(gate_key) { Entry.new(now, 0, now, 0) }
|
|
75
85
|
end
|
|
76
86
|
|
|
87
|
+
# Drop entries whose minute window has ended. At most once per window:
|
|
88
|
+
# while the map is full every unseen key lands here, and a full scan per
|
|
89
|
+
# miss would put an O(n) walk on the capture path. Racy by design (two
|
|
90
|
+
# threads may both sweep); deleting an expired entry twice is harmless.
|
|
91
|
+
def sweep_expired!(now)
|
|
92
|
+
return if @last_sweep && now - @last_sweep < WINDOW_SECONDS
|
|
93
|
+
|
|
94
|
+
@last_sweep = now
|
|
95
|
+
@entries.each_pair do |key, entry|
|
|
96
|
+
@entries.delete(key) if now - entry.window_start >= WINDOW_SECONDS
|
|
97
|
+
end
|
|
98
|
+
end
|
|
99
|
+
|
|
77
100
|
def roll_windows(entry, now)
|
|
78
101
|
if now - entry.window_start >= WINDOW_SECONDS
|
|
79
102
|
entry.window_start = now
|
|
@@ -108,6 +108,11 @@ module RailsErrorDashboard
|
|
|
108
108
|
def flush_if_due!
|
|
109
109
|
return unless enabled?
|
|
110
110
|
|
|
111
|
+
# Advance the breaker by the clock before flushing. When errors
|
|
112
|
+
# stop, this (end of every request and job) is the only thing left
|
|
113
|
+
# that can move it out of :open, and doing it first means an
|
|
114
|
+
# episode that has just ended is persisted by this very flush.
|
|
115
|
+
breaker.tick!
|
|
111
116
|
maybe_flush!
|
|
112
117
|
nil
|
|
113
118
|
rescue => e
|
|
@@ -159,6 +164,7 @@ module RailsErrorDashboard
|
|
|
159
164
|
@count_buffer = nil
|
|
160
165
|
@fingerprint_buckets = nil
|
|
161
166
|
@probe_counter = nil
|
|
167
|
+
@probe_epoch = nil
|
|
162
168
|
@issue_window_start = nil
|
|
163
169
|
@issue_window_count = nil
|
|
164
170
|
@last_flush = nil
|
|
@@ -180,8 +186,11 @@ module RailsErrorDashboard
|
|
|
180
186
|
:count_only
|
|
181
187
|
when :half_open
|
|
182
188
|
# Probe: a trickle of :lite captures tells us whether the storm
|
|
183
|
-
# has actually subsided; everything else stays counted.
|
|
184
|
-
|
|
189
|
+
# has actually subsided; everything else stays counted. The FIRST
|
|
190
|
+
# event of each half-open period is the probe (then every tenth):
|
|
191
|
+
# a recovering app with a slow trickle must not wait nine errors
|
|
192
|
+
# before we look at one.
|
|
193
|
+
if next_probe_index % 10 == 1
|
|
185
194
|
:lite
|
|
186
195
|
else
|
|
187
196
|
count!(exception, context)
|
|
@@ -238,7 +247,11 @@ module RailsErrorDashboard
|
|
|
238
247
|
def gate_parts(exception, context)
|
|
239
248
|
raw_message = exception.message.to_s[0, ErrorHashGenerator::HASH_MESSAGE_LIMIT]
|
|
240
249
|
|
|
241
|
-
|
|
250
|
+
# Everything below is buffered, JSON-encoded for StormFlushJob and
|
|
251
|
+
# later INSERTed, so no String may keep an invalid byte. The identity
|
|
252
|
+
# digest is computed from the raw message first (it is hex, and must
|
|
253
|
+
# match what the sync and async paths hash).
|
|
254
|
+
EncodingSanitizer.scrub_deep(
|
|
242
255
|
error_class: exception.class.name,
|
|
243
256
|
# The identity is hashed from the RAW message here, on the hot
|
|
244
257
|
# path, and only the digest is buffered. The message itself is
|
|
@@ -256,7 +269,8 @@ module RailsErrorDashboard
|
|
|
256
269
|
controller_name: context[:controller_name]&.to_s,
|
|
257
270
|
action_name: context[:action_name]&.to_s
|
|
258
271
|
),
|
|
259
|
-
|
|
272
|
+
# Scrubbed before redaction: the filter's regexes raise on invalid bytes.
|
|
273
|
+
message: redact(EncodingSanitizer.scrub(raw_message)),
|
|
260
274
|
first_app_frame: ErrorHashGenerator.extract_app_frame_from_locations(exception) ||
|
|
261
275
|
ErrorHashGenerator.extract_app_frame(exception.backtrace),
|
|
262
276
|
controller_name: context[:controller_name]&.to_s,
|
|
@@ -266,7 +280,7 @@ module RailsErrorDashboard
|
|
|
266
280
|
# worker's own environment", resolved at flush time exactly as
|
|
267
281
|
# LogError resolves it for a full capture.
|
|
268
282
|
environment: context[:environment].to_s.strip.presence&.[](0, 64)
|
|
269
|
-
|
|
283
|
+
)
|
|
270
284
|
end
|
|
271
285
|
|
|
272
286
|
# When a custom fingerprint lambda is configured the canonical hash
|
|
@@ -299,6 +313,19 @@ module RailsErrorDashboard
|
|
|
299
313
|
@probe_counter ||= Concurrent::AtomicFixnum.new(0)
|
|
300
314
|
end
|
|
301
315
|
|
|
316
|
+
# 1-based index of this event within the current half-open period.
|
|
317
|
+
# The reset is deliberately lock-free: two threads racing on a fresh
|
|
318
|
+
# epoch can at worst both start from a new counter, which admits one
|
|
319
|
+
# extra :lite probe. Nothing is lost and nothing can raise.
|
|
320
|
+
def next_probe_index
|
|
321
|
+
epoch = breaker.half_open_epoch
|
|
322
|
+
if @probe_epoch != epoch
|
|
323
|
+
@probe_epoch = epoch
|
|
324
|
+
@probe_counter = Concurrent::AtomicFixnum.new(0)
|
|
325
|
+
end
|
|
326
|
+
probe_counter.increment
|
|
327
|
+
end
|
|
328
|
+
|
|
302
329
|
# Piggyback flush (SwallowedExceptionTracker pattern): cheap
|
|
303
330
|
# timestamp check per admit; enqueue at most once per interval.
|
|
304
331
|
def maybe_flush!
|