rails_error_dashboard 0.12.1 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (90) hide show
  1. checksums.yaml +4 -4
  2. data/app/controllers/rails_error_dashboard/application_controller.rb +92 -11
  3. data/app/controllers/rails_error_dashboard/errors_controller.rb +111 -34
  4. data/app/controllers/rails_error_dashboard/webhooks_controller.rb +3 -0
  5. data/app/helpers/rails_error_dashboard/application_helper.rb +46 -1
  6. data/app/helpers/rails_error_dashboard/backtrace_helper.rb +8 -0
  7. data/app/jobs/rails_error_dashboard/concerns/plain_channel_message.rb +71 -0
  8. data/app/jobs/rails_error_dashboard/notification_burst_summary_job.rb +55 -0
  9. data/app/jobs/rails_error_dashboard/retention_cleanup_job.rb +68 -1
  10. data/app/jobs/rails_error_dashboard/storm_notification_job.rb +8 -44
  11. data/app/models/rails_error_dashboard/error_baseline.rb +12 -9
  12. data/app/models/rails_error_dashboard/error_comment.rb +0 -5
  13. data/app/models/rails_error_dashboard/error_log.rb +9 -3
  14. data/app/models/rails_error_dashboard/error_logs_record.rb +34 -0
  15. data/app/models/rails_error_dashboard/error_occurrence.rb +9 -1
  16. data/app/views/layouts/rails_error_dashboard.html.erb +11 -3
  17. data/app/views/rails_error_dashboard/errors/_discussion.html.erb +18 -11
  18. data/app/views/rails_error_dashboard/errors/_issue_section.html.erb +22 -8
  19. data/app/views/rails_error_dashboard/errors/_sidebar_metadata.html.erb +15 -6
  20. data/app/views/rails_error_dashboard/errors/analytics.html.erb +8 -8
  21. data/app/views/rails_error_dashboard/errors/diagnostic_dumps.html.erb +1 -1
  22. data/app/views/rails_error_dashboard/errors/index.html.erb +6 -1
  23. data/app/views/rails_error_dashboard/errors/platform_comparison.html.erb +7 -7
  24. data/app/views/rails_error_dashboard/errors/releases.html.erb +2 -2
  25. data/app/views/rails_error_dashboard/errors/settings.html.erb +2 -0
  26. data/config/locales/de.yml +28 -0
  27. data/config/locales/en.yml +35 -0
  28. data/config/locales/es.yml +28 -0
  29. data/config/locales/fr.yml +28 -0
  30. data/config/locales/it.yml +28 -0
  31. data/config/locales/ja.yml +28 -0
  32. data/config/locales/pl.yml +28 -0
  33. data/config/locales/pt-BR.yml +28 -0
  34. data/config/locales/ru.yml +28 -0
  35. data/config/locales/uk.yml +28 -0
  36. data/config/locales/zh-CN.yml +28 -0
  37. data/db/migrate/20260917000001_add_last_notified_at_to_error_logs.rb +40 -0
  38. data/lib/generators/rails_error_dashboard/install/templates/initializer.rb +12 -1
  39. data/lib/rails_error_dashboard/commands/assign_error.rb +12 -2
  40. data/lib/rails_error_dashboard/commands/backfill_environments.rb +2 -0
  41. data/lib/rails_error_dashboard/commands/backfill_resolved_at.rb +42 -0
  42. data/lib/rails_error_dashboard/commands/batch_delete_errors.rb +1 -0
  43. data/lib/rails_error_dashboard/commands/batch_mute_errors.rb +2 -0
  44. data/lib/rails_error_dashboard/commands/batch_resolve_errors.rb +2 -0
  45. data/lib/rails_error_dashboard/commands/batch_unmute_errors.rb +2 -0
  46. data/lib/rails_error_dashboard/commands/find_or_increment_error.rb +39 -7
  47. data/lib/rails_error_dashboard/commands/flush_rack_attack_events.rb +3 -1
  48. data/lib/rails_error_dashboard/commands/flush_storm_counts.rb +27 -4
  49. data/lib/rails_error_dashboard/commands/flush_swallowed_exceptions.rb +5 -2
  50. data/lib/rails_error_dashboard/commands/link_existing_issue.rb +1 -0
  51. data/lib/rails_error_dashboard/commands/log_error.rb +82 -23
  52. data/lib/rails_error_dashboard/commands/mute_error.rb +1 -0
  53. data/lib/rails_error_dashboard/commands/resolve_error.rb +2 -0
  54. data/lib/rails_error_dashboard/commands/scrub_invalid_encoding.rb +104 -0
  55. data/lib/rails_error_dashboard/commands/snooze_error.rb +34 -10
  56. data/lib/rails_error_dashboard/commands/unmute_error.rb +1 -0
  57. data/lib/rails_error_dashboard/commands/update_error_priority.rb +23 -2
  58. data/lib/rails_error_dashboard/commands/update_error_status.rb +33 -6
  59. data/lib/rails_error_dashboard/configuration.rb +19 -1
  60. data/lib/rails_error_dashboard/engine.rb +15 -0
  61. data/lib/rails_error_dashboard/queries/analytics_stats.rb +4 -3
  62. data/lib/rails_error_dashboard/queries/baseline_stats.rb +107 -0
  63. data/lib/rails_error_dashboard/queries/dashboard_stats.rb +55 -50
  64. data/lib/rails_error_dashboard/queries/error_correlation.rb +6 -3
  65. data/lib/rails_error_dashboard/queries/errors_list.rb +15 -2
  66. data/lib/rails_error_dashboard/queries/similar_errors.rb +1 -1
  67. data/lib/rails_error_dashboard/services/analytics_cache_manager.rb +43 -18
  68. data/lib/rails_error_dashboard/services/backtrace_processor.rb +3 -1
  69. data/lib/rails_error_dashboard/services/cause_chain_extractor.rb +3 -1
  70. data/lib/rails_error_dashboard/services/codeberg_issue_client.rb +13 -2
  71. data/lib/rails_error_dashboard/services/diagnostic_dump_generator.rb +5 -3
  72. data/lib/rails_error_dashboard/services/encoding_sanitizer.rb +80 -0
  73. data/lib/rails_error_dashboard/services/error_broadcaster.rb +180 -32
  74. data/lib/rails_error_dashboard/services/error_hash_generator.rb +9 -3
  75. data/lib/rails_error_dashboard/services/error_notification_dispatcher.rb +13 -0
  76. data/lib/rails_error_dashboard/services/exception_filter.rb +55 -0
  77. data/lib/rails_error_dashboard/services/git_head_reader.rb +102 -0
  78. data/lib/rails_error_dashboard/services/notification_throttler.rb +172 -28
  79. data/lib/rails_error_dashboard/services/sensitive_data_filter.rb +47 -1
  80. data/lib/rails_error_dashboard/services/storm_protection/circuit_breaker.rb +59 -3
  81. data/lib/rails_error_dashboard/services/storm_protection/fingerprint_buckets.rb +25 -2
  82. data/lib/rails_error_dashboard/services/storm_protection/gate.rb +32 -5
  83. data/lib/rails_error_dashboard/services/swallowed_exception_tracker.rb +99 -23
  84. data/lib/rails_error_dashboard/services/url_safety.rb +40 -0
  85. data/lib/rails_error_dashboard/subscribers/issue_tracker_subscriber.rb +11 -2
  86. data/lib/rails_error_dashboard/value_objects/error_context.rb +3 -1
  87. data/lib/rails_error_dashboard/version.rb +1 -1
  88. data/lib/rails_error_dashboard.rb +31 -0
  89. data/lib/tasks/error_dashboard.rake +54 -4
  90. metadata +10 -2
@@ -2,21 +2,73 @@
2
2
 
3
3
  module RailsErrorDashboard
4
4
  module Services
5
- # Pure algorithm: Throttle error notifications to prevent alert fatigue
5
+ # Throttle error notifications to prevent alert fatigue
6
6
  #
7
7
  # Checks severity minimum, per-error cooldown, and threshold milestones.
8
- # Uses in-memory cache (same pattern as BaselineAlertThrottler).
8
+ #
9
+ # The cooldown is CLAIMED IN THE DATABASE (error_logs.last_notified_at, see
10
+ # claim!). A per-process Hash cannot do this job: every Puma worker and
11
+ # every job process has its own, so one error reopened by a bad deploy
12
+ # notified once per process, and a restart forgot the cooldown altogether.
13
+ # The Hash survives only as a bounded fallback for the window between
14
+ # upgrading the gem and running its migration, and for callers that pass
15
+ # something other than a persisted row.
16
+ #
9
17
  # Thread-safe via Mutex. Fail-open: returns true on any error.
10
18
  class NotificationThrottler
11
19
  # Severity levels ranked from lowest to highest
12
20
  SEVERITY_RANK = { low: 0, medium: 1, high: 2, critical: 3 }.freeze
13
21
 
22
+ # Hard cap on the in-process fallback. Expired entries are swept on every
23
+ # insert; the cap only bites when more than this many distinct errors are
24
+ # inside their cooldown at once, and then the oldest goes first.
25
+ MAX_TRACKED = 1_000
26
+
27
+ COOLDOWN_COLUMN = "last_notified_at"
28
+
14
29
  @last_notification_times = {}
15
30
  @mutex = Mutex.new
31
+ @burst_mutex = Mutex.new
32
+ @burst_window_start = nil
33
+ @burst_count = 0
16
34
 
17
35
  class << self
18
- # Should we send a notification for this error?
19
- # Checks: severity minimum + cooldown period
36
+ # Take the right to notify about this error, atomically.
37
+ #
38
+ # Call it immediately BEFORE dispatching, and dispatch only on true. It
39
+ # is claim-then-send: a claim followed by a failed send is not handed
40
+ # back, so that error stays quiet until the cooldown ends. The
41
+ # alternative (send, then record) is what let N processes all send.
42
+ #
43
+ # With the column present this is ONE statement and no read:
44
+ #
45
+ # UPDATE error_logs SET last_notified_at = :now
46
+ # WHERE id = :id AND (last_notified_at IS NULL OR last_notified_at < :cutoff)
47
+ #
48
+ # Exactly one connection can match the row, on SQLite, PostgreSQL and
49
+ # MySQL alike, so exactly one process notifies per cooldown window.
50
+ #
51
+ # @param error_log [ErrorLog] the row being notified about
52
+ # @param respect_cooldown [Boolean] false for notifications the cooldown
53
+ # has never applied to (first occurrence, threshold milestones): always
54
+ # granted, but still stamped, so a reopen soon afterwards is throttled
55
+ # @return [Boolean] true if the caller may notify
56
+ def claim!(error_log, respect_cooldown: true)
57
+ minutes = respect_cooldown ? cooldown_minutes : 0
58
+
59
+ if database_claim?(error_log)
60
+ claim_in_database(error_log, minutes)
61
+ else
62
+ claim_in_process(error_log, minutes)
63
+ end
64
+ rescue => e
65
+ # Fail-open: a throttler that cannot decide must not lose a page.
66
+ RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] NotificationThrottler.claim! failed: #{e.class}: #{e.message}")
67
+ true
68
+ end
69
+
70
+ # Should we send a notification for this error? Read-only (severity
71
+ # minimum + cooldown): it takes nothing. claim! is what notifies.
20
72
  # @param error_log [ErrorLog] The error to check
21
73
  # @return [Boolean] true if notification should be sent
22
74
  def should_notify?(error_log)
@@ -79,53 +131,145 @@ module RailsErrorDashboard
79
131
  false
80
132
  end
81
133
 
82
- # Record that a notification was sent for this error
83
- # @param error_log [ErrorLog] The error that was notified about
84
- def record_notification(error_log)
85
- key = error_log.error_hash
134
+ # May a FIRST-OCCURRENCE notification go out right now?
135
+ #
136
+ # The per-error cooldown cannot bound a bad deploy that produces hundreds
137
+ # of DISTINCT new errors: each is a first occurrence, and first
138
+ # occurrences always notify. This is a fixed window counter:
139
+ #
140
+ # :notify within config.notification_burst_limit for this window
141
+ # :summarize the first one over the limit: send ONE summary instead
142
+ # :suppress everything after that, until the window ends
143
+ #
144
+ # Per process, like Gate.issue_creation_allowed?, because there is no
145
+ # store every deployment shares except the database and this is asked on
146
+ # the capture path. Worst case is limit x processes per window.
147
+ #
148
+ # Each call consumes a slot: ask only when about to notify. A limit or
149
+ # window of 0 / nil turns the cap off. Fails open to :notify.
150
+ #
151
+ # @return [Symbol] :notify, :summarize or :suppress
152
+ def burst_decision
153
+ limit = RailsErrorDashboard.configuration.notification_burst_limit.to_i
154
+ window = RailsErrorDashboard.configuration.notification_burst_window_seconds.to_i
155
+ return :notify unless limit.positive? && window.positive?
86
156
 
87
- @mutex.synchronize do
88
- @last_notification_times[key] = Time.current
157
+ now = monotonic_now
158
+ count = @burst_mutex.synchronize do
159
+ if @burst_window_start.nil? || now - @burst_window_start >= window
160
+ @burst_window_start = now
161
+ @burst_count = 0
162
+ end
163
+ @burst_count += 1
164
+ end
165
+
166
+ if count <= limit
167
+ :notify
168
+ elsif count == limit + 1
169
+ :summarize
170
+ else
171
+ :suppress
89
172
  end
90
173
  rescue => e
91
- RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] NotificationThrottler.record_notification failed: #{e.message}")
174
+ RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] NotificationThrottler.burst_decision failed: #{e.class}: #{e.message}")
175
+ :notify
176
+ end
177
+
178
+ # Record that a notification was sent for this error, unconditionally.
179
+ # Kept for callers that notify outside LogError; LogError itself uses
180
+ # claim!, which decides and records in one step.
181
+ # @param error_log [ErrorLog] The error that was notified about
182
+ def record_notification(error_log)
183
+ claim!(error_log, respect_cooldown: false)
184
+ nil
92
185
  end
93
186
 
94
- # Clear all throttle state (for testing)
187
+ # Clear all in-process throttle state (for testing)
95
188
  def clear!
96
189
  @mutex.synchronize do
97
190
  @last_notification_times.clear
98
191
  end
192
+ @burst_mutex.synchronize do
193
+ @burst_window_start = nil
194
+ @burst_count = 0
195
+ end
99
196
  end
100
197
 
101
- # Remove old entries to prevent memory growth
102
- # @param max_age_hours [Integer] Remove entries older than this (default: 24)
103
- def cleanup!(max_age_hours: 24)
104
- cutoff_time = max_age_hours.hours.ago
198
+ private
105
199
 
106
- @mutex.synchronize do
107
- @last_notification_times.delete_if { |_, time| time < cutoff_time }
108
- end
200
+ def monotonic_now
201
+ Process.clock_gettime(Process::CLOCK_MONOTONIC)
109
202
  end
110
203
 
111
- private
204
+ def cooldown_minutes
205
+ RailsErrorDashboard.configuration.notification_cooldown_minutes.to_i
206
+ end
112
207
 
113
- # Is the error outside the cooldown window?
114
- # @param error_log [ErrorLog] The error to check
115
- # @return [Boolean] true if not in cooldown (ok to notify)
116
- def cooldown_ok?(error_log)
117
- cooldown_minutes = RailsErrorDashboard.configuration.notification_cooldown_minutes
118
- return true if cooldown_minutes.nil? || cooldown_minutes <= 0
208
+ # A persisted row AND the column: before the migration has run the
209
+ # UPDATE would raise on every notification.
210
+ def database_claim?(error_log)
211
+ error_log.is_a?(ErrorLog) && error_log.persisted? &&
212
+ ErrorLog.column_names.include?(COOLDOWN_COLUMN)
213
+ end
119
214
 
215
+ def claim_in_database(error_log, minutes)
216
+ now = Time.current
217
+ scope = ErrorLog.where(id: error_log.id)
218
+ if minutes.positive?
219
+ scope = scope.where("#{COOLDOWN_COLUMN} IS NULL OR #{COOLDOWN_COLUMN} < ?", now - minutes.minutes)
220
+ end
221
+
222
+ granted = scope.update_all(COOLDOWN_COLUMN => now) == 1
223
+ # No cooldown to lose: a row deleted under us must not silence the page.
224
+ granted || !minutes.positive?
225
+ end
226
+
227
+ def claim_in_process(error_log, minutes)
120
228
  key = error_log.error_hash
229
+ now = Time.current
121
230
 
122
231
  @mutex.synchronize do
123
232
  last_time = @last_notification_times[key]
124
- return true if last_time.nil?
233
+ next false if minutes.positive? && last_time && now <= last_time + minutes.minutes
125
234
 
126
- Time.current > (last_time + cooldown_minutes.minutes)
235
+ remember(key, now)
236
+ true
127
237
  end
128
238
  end
239
+
240
+ # Caller holds @mutex. Delete-and-reinsert keeps the Hash in recency
241
+ # order, so "oldest" is simply the first key.
242
+ def remember(key, now)
243
+ window = cooldown_minutes
244
+ @last_notification_times.delete(key)
245
+ # With no cooldown nothing will ever read the entry back.
246
+ return unless window.positive?
247
+
248
+ cutoff = now - window.minutes
249
+ @last_notification_times.delete_if { |_, time| time < cutoff }
250
+ @last_notification_times.shift while @last_notification_times.size >= MAX_TRACKED
251
+ @last_notification_times[key] = now
252
+ end
253
+
254
+ # Is the error outside the cooldown window? Read-only.
255
+ # @param error_log [ErrorLog] The error to check
256
+ # @return [Boolean] true if not in cooldown (ok to notify)
257
+ def cooldown_ok?(error_log)
258
+ minutes = cooldown_minutes
259
+ return true unless minutes.positive?
260
+
261
+ last_time =
262
+ if database_claim?(error_log)
263
+ # Read the ROW, not the object: claim! stamps it with update_all,
264
+ # so the caller's copy -- and every other process's -- is stale.
265
+ ErrorLog.where(id: error_log.id).pick(COOLDOWN_COLUMN)
266
+ else
267
+ @mutex.synchronize { @last_notification_times[error_log.error_hash] }
268
+ end
269
+ return true if last_time.nil?
270
+
271
+ Time.current > (last_time + minutes.minutes)
272
+ end
129
273
  end
130
274
  end
131
275
  end
@@ -51,7 +51,53 @@ module RailsErrorDashboard
51
51
  attributes
52
52
  end
53
53
 
54
- # Build and cache the ParameterFilter instance
54
+ SESSION_DIGEST_PREFIX = "h1:"
55
+ SESSION_DIGEST_FORMAT = /\Ah1:\h{32}\z/
56
+
57
+ # Keyed digest of a session ID, for storage.
58
+ #
59
+ # A session ID is a bearer credential, and the only thing the dashboard does
60
+ # with it is equality ("these occurrences were one session"). An HMAC keeps
61
+ # that and is useless for replay; keying it with secret_key_base means a
62
+ # guessed ID cannot be confirmed offline either. The prefix tells a digest
63
+ # from a raw Rails session ID (itself 32 hex characters), which makes
64
+ # digesting idempotent and leaves room for an "h2:".
65
+ #
66
+ # Rotating secret_key_base breaks correlation across the rotation.
67
+ #
68
+ # @param raw [#to_s, nil] raw session ID, or an existing digest
69
+ # @return [String, nil] "h1:" + 32 hex characters; nil for blank input or on
70
+ # any failure (never the raw value, never an exception)
71
+ def self.digest_session_id(raw)
72
+ value = raw.to_s
73
+ return nil if value.empty?
74
+ # Exact shape only: a raw value that merely starts with "h1:" is digested.
75
+ return value if value.match?(SESSION_DIGEST_FORMAT)
76
+
77
+ SESSION_DIGEST_PREFIX + OpenSSL::HMAC.hexdigest("SHA256", session_digest_key, value)[0, 32]
78
+ rescue => e
79
+ RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] Session ID digest failed: #{e.class}")
80
+ nil
81
+ end
82
+
83
+ # What to write to error_occurrences.session_id: the digest while filtering
84
+ # is on, the raw value when the operator has turned filtering off.
85
+ def self.storable_session_id(raw)
86
+ return digest_session_id(raw) if RailsErrorDashboard.configuration.filter_sensitive_data
87
+
88
+ raw.nil? ? nil : raw.to_s
89
+ rescue => e
90
+ RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] Session ID handling failed: #{e.class}")
91
+ nil
92
+ end
93
+
94
+ def self.session_digest_key
95
+ key = Rails.application.secret_key_base if defined?(Rails) && Rails.application.respond_to?(:secret_key_base)
96
+ key.to_s.empty? ? "rails_error_dashboard" : key.to_s
97
+ rescue StandardError
98
+ "rails_error_dashboard"
99
+ end
100
+
55
101
  # @return [ActiveSupport::ParameterFilter, nil]
56
102
  def self.parameter_filter
57
103
  @parameter_filter ||= build_parameter_filter
@@ -22,8 +22,15 @@ module RailsErrorDashboard
22
22
  class CircuitBreaker
23
23
  BUCKET_SECONDS = 10
24
24
  CALM_BUCKETS_TO_CLOSE = 2
25
+ # Empty buckets replayed after a silence. Seven is enough to walk
26
+ # :open -> :half_open -> :closed; the cap keeps a roll O(1) however long
27
+ # the process sat idle.
28
+ MAX_CATCH_UP_BUCKETS = 7
25
29
 
26
- attr_reader :state
30
+ # Incremented every time the breaker ENTERS :half_open. The gate compares
31
+ # it with the last value it saw to restart its probe counter, so each
32
+ # recovery attempt probes with its first event.
33
+ attr_reader :half_open_epoch
27
34
 
28
35
  # @param clock [#call] returns monotonic seconds; injectable for tests
29
36
  def initialize(clock: -> { Process.clock_gettime(Process::CLOCK_MONOTONIC) })
@@ -40,6 +47,7 @@ module RailsErrorDashboard
40
47
  @calm_buckets = 0
41
48
  @opened_at = nil
42
49
  @episode = nil
50
+ @half_open_epoch = 0
43
51
  end
44
52
  end
45
53
 
@@ -61,6 +69,25 @@ module RailsErrorDashboard
61
69
  @state
62
70
  end
63
71
 
72
+ # Current state, advanced by elapsed TIME as well as by events. Without the
73
+ # tick a storm that simply stopped left the breaker open for ever: only
74
+ # record! rolled buckets, and nothing calls record! when errors stop.
75
+ def state
76
+ tick!
77
+ @state
78
+ end
79
+
80
+ # Roll the bucket if one is due. Cheap when it is not: one clock read and
81
+ # a comparison, no lock. Safe to call from anywhere, any number of times.
82
+ def tick!
83
+ now = @clock.call
84
+ roll!(now) if now - @bucket_start >= BUCKET_SECONDS
85
+ nil
86
+ rescue => e
87
+ RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] CircuitBreaker#tick! failed: #{e.class}: #{e.message}")
88
+ nil
89
+ end
90
+
64
91
  # Episode metadata for the honesty layer (storm_events row).
65
92
  # @return [Hash, nil] nil when no episode is active or recently closed
66
93
  def episode_snapshot
@@ -81,10 +108,38 @@ module RailsErrorDashboard
81
108
  elapsed = now - @bucket_start
82
109
  return if elapsed < BUCKET_SECONDS # another thread already rolled
83
110
 
84
- rate = @bucket_count.value / elapsed.to_f
111
+ count = @bucket_count.value
112
+ bucket_start = @bucket_start
85
113
  @bucket_start = now
86
114
  @bucket_count = Concurrent::AtomicFixnum.new(0)
87
- transition!(rate, now)
115
+
116
+ buckets = (elapsed / BUCKET_SECONDS).floor
117
+ if buckets <= 1
118
+ transition!(count / elapsed.to_f, now)
119
+ else
120
+ catch_up!(count, elapsed, bucket_start, now, buckets)
121
+ end
122
+ end
123
+ end
124
+
125
+ # More than one bucket of wall time has passed since the last roll, so
126
+ # nobody called in between: replay the gap instead of treating it as one
127
+ # long bucket. Each step carries the time its bucket ENDED, which is what
128
+ # keeps the cooldown and the two-calm-bucket rule honest (10s past the
129
+ # cooldown is :half_open, not :closed).
130
+ #
131
+ # The measured bucket keeps the diluted rate (count / elapsed) it has
132
+ # always had, so a burst followed by silence never escalates a closed
133
+ # breaker after the fact. Only the most recent MAX_CATCH_UP_BUCKETS empty
134
+ # buckets are replayed; older ones could not change the outcome.
135
+ def catch_up!(count, elapsed, bucket_start, now, buckets)
136
+ transition!(count / elapsed.to_f, bucket_start + BUCKET_SECONDS)
137
+
138
+ empty = [ buckets - 1, MAX_CATCH_UP_BUCKETS ].min
139
+ (empty - 1).downto(0) do |back|
140
+ break if @state == :closed
141
+
142
+ transition!(0.0, now - (back * BUCKET_SECONDS))
88
143
  end
89
144
  end
90
145
 
@@ -111,6 +166,7 @@ module RailsErrorDashboard
111
166
  if now - @opened_at >= cooldown_seconds && rate < shedding_threshold
112
167
  @state = :half_open
113
168
  @calm_buckets = 0
169
+ @half_open_epoch += 1
114
170
  end
115
171
  when :half_open
116
172
  if rate >= shedding_threshold
@@ -39,6 +39,7 @@ module RailsErrorDashboard
39
39
 
40
40
  def reset!
41
41
  @entries = Concurrent::Map.new
42
+ @last_sweep = nil
42
43
  end
43
44
 
44
45
  # Decide capture fidelity for one event of this fingerprint.
@@ -68,12 +69,34 @@ module RailsErrorDashboard
68
69
  return existing if existing
69
70
 
70
71
  # Bounded: never insert past the cap (size check is approximate
71
- # under concurrency — a few entries over the cap is fine)
72
- return nil if @entries.size >= max_tracked
72
+ # under concurrency — a few entries over the cap is fine). Before
73
+ # declining, make room by dropping fingerprints that have gone quiet:
74
+ # without that the map filled once and stayed full for the life of
75
+ # the process, so the first N fingerprints a worker ever saw were the
76
+ # only ones it would ever rate-limit. If every tracked entry is still
77
+ # live, a new key is still declined — a storm of unique fingerprints
78
+ # is the global breaker's job (Layer 2), not this map's.
79
+ if @entries.size >= max_tracked
80
+ sweep_expired!(now)
81
+ return nil if @entries.size >= max_tracked
82
+ end
73
83
 
74
84
  @entries.compute_if_absent(gate_key) { Entry.new(now, 0, now, 0) }
75
85
  end
76
86
 
87
+ # Drop entries whose minute window has ended. At most once per window:
88
+ # while the map is full every unseen key lands here, and a full scan per
89
+ # miss would put an O(n) walk on the capture path. Racy by design (two
90
+ # threads may both sweep); deleting an expired entry twice is harmless.
91
+ def sweep_expired!(now)
92
+ return if @last_sweep && now - @last_sweep < WINDOW_SECONDS
93
+
94
+ @last_sweep = now
95
+ @entries.each_pair do |key, entry|
96
+ @entries.delete(key) if now - entry.window_start >= WINDOW_SECONDS
97
+ end
98
+ end
99
+
77
100
  def roll_windows(entry, now)
78
101
  if now - entry.window_start >= WINDOW_SECONDS
79
102
  entry.window_start = now
@@ -108,6 +108,11 @@ module RailsErrorDashboard
108
108
  def flush_if_due!
109
109
  return unless enabled?
110
110
 
111
+ # Advance the breaker by the clock before flushing. When errors
112
+ # stop, this (end of every request and job) is the only thing left
113
+ # that can move it out of :open, and doing it first means an
114
+ # episode that has just ended is persisted by this very flush.
115
+ breaker.tick!
111
116
  maybe_flush!
112
117
  nil
113
118
  rescue => e
@@ -159,6 +164,7 @@ module RailsErrorDashboard
159
164
  @count_buffer = nil
160
165
  @fingerprint_buckets = nil
161
166
  @probe_counter = nil
167
+ @probe_epoch = nil
162
168
  @issue_window_start = nil
163
169
  @issue_window_count = nil
164
170
  @last_flush = nil
@@ -180,8 +186,11 @@ module RailsErrorDashboard
180
186
  :count_only
181
187
  when :half_open
182
188
  # Probe: a trickle of :lite captures tells us whether the storm
183
- # has actually subsided; everything else stays counted.
184
- if (probe_counter.increment % 10).zero?
189
+ # has actually subsided; everything else stays counted. The FIRST
190
+ # event of each half-open period is the probe (then every tenth):
191
+ # a recovering app with a slow trickle must not wait nine errors
192
+ # before we look at one.
193
+ if next_probe_index % 10 == 1
185
194
  :lite
186
195
  else
187
196
  count!(exception, context)
@@ -238,7 +247,11 @@ module RailsErrorDashboard
238
247
  def gate_parts(exception, context)
239
248
  raw_message = exception.message.to_s[0, ErrorHashGenerator::HASH_MESSAGE_LIMIT]
240
249
 
241
- {
250
+ # Everything below is buffered, JSON-encoded for StormFlushJob and
251
+ # later INSERTed, so no String may keep an invalid byte. The identity
252
+ # digest is computed from the raw message first (it is hex, and must
253
+ # match what the sync and async paths hash).
254
+ EncodingSanitizer.scrub_deep(
242
255
  error_class: exception.class.name,
243
256
  # The identity is hashed from the RAW message here, on the hot
244
257
  # path, and only the digest is buffered. The message itself is
@@ -256,7 +269,8 @@ module RailsErrorDashboard
256
269
  controller_name: context[:controller_name]&.to_s,
257
270
  action_name: context[:action_name]&.to_s
258
271
  ),
259
- message: redact(raw_message),
272
+ # Scrubbed before redaction: the filter's regexes raise on invalid bytes.
273
+ message: redact(EncodingSanitizer.scrub(raw_message)),
260
274
  first_app_frame: ErrorHashGenerator.extract_app_frame_from_locations(exception) ||
261
275
  ErrorHashGenerator.extract_app_frame(exception.backtrace),
262
276
  controller_name: context[:controller_name]&.to_s,
@@ -266,7 +280,7 @@ module RailsErrorDashboard
266
280
  # worker's own environment", resolved at flush time exactly as
267
281
  # LogError resolves it for a full capture.
268
282
  environment: context[:environment].to_s.strip.presence&.[](0, 64)
269
- }
283
+ )
270
284
  end
271
285
 
272
286
  # When a custom fingerprint lambda is configured the canonical hash
@@ -299,6 +313,19 @@ module RailsErrorDashboard
299
313
  @probe_counter ||= Concurrent::AtomicFixnum.new(0)
300
314
  end
301
315
 
316
+ # 1-based index of this event within the current half-open period.
317
+ # The reset is deliberately lock-free: two threads racing on a fresh
318
+ # epoch can at worst both start from a new counter, which admits one
319
+ # extra :lite probe. Nothing is lost and nothing can raise.
320
+ def next_probe_index
321
+ epoch = breaker.half_open_epoch
322
+ if @probe_epoch != epoch
323
+ @probe_epoch = epoch
324
+ @probe_counter = Concurrent::AtomicFixnum.new(0)
325
+ end
326
+ probe_counter.increment
327
+ end
328
+
302
329
  # Piggyback flush (SwallowedExceptionTracker pattern): cheap
303
330
  # timestamp check per admit; enqueue at most once per interval.
304
331
  def maybe_flush!