rails_error_dashboard 0.12.1 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. checksums.yaml +4 -4
  2. data/app/controllers/rails_error_dashboard/application_controller.rb +92 -11
  3. data/app/controllers/rails_error_dashboard/errors_controller.rb +111 -34
  4. data/app/controllers/rails_error_dashboard/webhooks_controller.rb +3 -0
  5. data/app/helpers/rails_error_dashboard/application_helper.rb +46 -1
  6. data/app/helpers/rails_error_dashboard/backtrace_helper.rb +8 -0
  7. data/app/jobs/rails_error_dashboard/async_error_logging_job.rb +10 -0
  8. data/app/jobs/rails_error_dashboard/concerns/plain_channel_message.rb +71 -0
  9. data/app/jobs/rails_error_dashboard/notification_burst_summary_job.rb +55 -0
  10. data/app/jobs/rails_error_dashboard/retention_cleanup_job.rb +122 -1
  11. data/app/jobs/rails_error_dashboard/storm_flush_job.rb +7 -4
  12. data/app/jobs/rails_error_dashboard/storm_notification_job.rb +8 -44
  13. data/app/models/rails_error_dashboard/error_baseline.rb +12 -9
  14. data/app/models/rails_error_dashboard/error_comment.rb +0 -5
  15. data/app/models/rails_error_dashboard/error_log.rb +19 -3
  16. data/app/models/rails_error_dashboard/error_logs_record.rb +34 -0
  17. data/app/models/rails_error_dashboard/error_occurrence.rb +9 -1
  18. data/app/models/rails_error_dashboard/event_count.rb +132 -0
  19. data/app/models/rails_error_dashboard/event_timing_gap.rb +55 -0
  20. data/app/views/layouts/rails_error_dashboard.html.erb +58 -5
  21. data/app/views/rails_error_dashboard/errors/_discussion.html.erb +18 -11
  22. data/app/views/rails_error_dashboard/errors/_issue_section.html.erb +22 -8
  23. data/app/views/rails_error_dashboard/errors/_request_context.html.erb +2 -0
  24. data/app/views/rails_error_dashboard/errors/_sidebar_metadata.html.erb +15 -6
  25. data/app/views/rails_error_dashboard/errors/analytics.html.erb +8 -8
  26. data/app/views/rails_error_dashboard/errors/diagnostic_dumps.html.erb +1 -1
  27. data/app/views/rails_error_dashboard/errors/index.html.erb +6 -1
  28. data/app/views/rails_error_dashboard/errors/overview.html.erb +12 -0
  29. data/app/views/rails_error_dashboard/errors/platform_comparison.html.erb +7 -7
  30. data/app/views/rails_error_dashboard/errors/releases.html.erb +2 -2
  31. data/app/views/rails_error_dashboard/errors/settings.html.erb +2 -0
  32. data/app/views/rails_error_dashboard/errors/show.html.erb +3 -3
  33. data/config/locales/de.yml +30 -0
  34. data/config/locales/en.yml +37 -0
  35. data/config/locales/es.yml +30 -0
  36. data/config/locales/fr.yml +30 -0
  37. data/config/locales/it.yml +30 -0
  38. data/config/locales/ja.yml +30 -0
  39. data/config/locales/pl.yml +30 -0
  40. data/config/locales/pt-BR.yml +30 -0
  41. data/config/locales/ru.yml +30 -0
  42. data/config/locales/uk.yml +30 -0
  43. data/config/locales/zh-CN.yml +30 -0
  44. data/db/migrate/20260917000001_add_last_notified_at_to_error_logs.rb +40 -0
  45. data/db/migrate/20260919000001_create_event_counts.rb +71 -0
  46. data/db/migrate/20260920000001_add_buckets_incomplete_to_storm_events.rb +25 -0
  47. data/db/migrate/20260920000002_create_event_timing_gaps.rb +55 -0
  48. data/lib/generators/rails_error_dashboard/install/templates/initializer.rb +12 -1
  49. data/lib/rails_error_dashboard/commands/assign_error.rb +12 -2
  50. data/lib/rails_error_dashboard/commands/backfill_environments.rb +2 -0
  51. data/lib/rails_error_dashboard/commands/backfill_resolved_at.rb +42 -0
  52. data/lib/rails_error_dashboard/commands/batch_delete_errors.rb +1 -0
  53. data/lib/rails_error_dashboard/commands/batch_mute_errors.rb +2 -0
  54. data/lib/rails_error_dashboard/commands/batch_resolve_errors.rb +2 -0
  55. data/lib/rails_error_dashboard/commands/batch_unmute_errors.rb +2 -0
  56. data/lib/rails_error_dashboard/commands/find_or_increment_error.rb +109 -11
  57. data/lib/rails_error_dashboard/commands/flush_rack_attack_events.rb +3 -1
  58. data/lib/rails_error_dashboard/commands/flush_storm_counts.rb +252 -12
  59. data/lib/rails_error_dashboard/commands/flush_swallowed_exceptions.rb +5 -2
  60. data/lib/rails_error_dashboard/commands/link_existing_issue.rb +1 -0
  61. data/lib/rails_error_dashboard/commands/log_error.rb +291 -40
  62. data/lib/rails_error_dashboard/commands/mute_error.rb +1 -0
  63. data/lib/rails_error_dashboard/commands/resolve_error.rb +2 -0
  64. data/lib/rails_error_dashboard/commands/scrub_invalid_encoding.rb +104 -0
  65. data/lib/rails_error_dashboard/commands/snooze_error.rb +34 -10
  66. data/lib/rails_error_dashboard/commands/unmute_error.rb +1 -0
  67. data/lib/rails_error_dashboard/commands/update_error_priority.rb +23 -2
  68. data/lib/rails_error_dashboard/commands/update_error_status.rb +33 -6
  69. data/lib/rails_error_dashboard/configuration.rb +39 -1
  70. data/lib/rails_error_dashboard/engine.rb +28 -0
  71. data/lib/rails_error_dashboard/manual_error_reporter.rb +16 -5
  72. data/lib/rails_error_dashboard/queries/analytics_stats.rb +89 -30
  73. data/lib/rails_error_dashboard/queries/baseline_stats.rb +107 -0
  74. data/lib/rails_error_dashboard/queries/dashboard_stats.rb +216 -74
  75. data/lib/rails_error_dashboard/queries/error_correlation.rb +6 -3
  76. data/lib/rails_error_dashboard/queries/errors_list.rb +15 -2
  77. data/lib/rails_error_dashboard/queries/event_volume.rb +503 -0
  78. data/lib/rails_error_dashboard/queries/similar_errors.rb +1 -1
  79. data/lib/rails_error_dashboard/services/analytics_cache_manager.rb +43 -18
  80. data/lib/rails_error_dashboard/services/backtrace_processor.rb +3 -1
  81. data/lib/rails_error_dashboard/services/breadcrumb_collector.rb +23 -0
  82. data/lib/rails_error_dashboard/services/cause_chain_extractor.rb +3 -1
  83. data/lib/rails_error_dashboard/services/codeberg_issue_client.rb +13 -2
  84. data/lib/rails_error_dashboard/services/diagnostic_dump_generator.rb +5 -3
  85. data/lib/rails_error_dashboard/services/encoding_sanitizer.rb +80 -0
  86. data/lib/rails_error_dashboard/services/error_broadcaster.rb +180 -32
  87. data/lib/rails_error_dashboard/services/error_hash_generator.rb +9 -3
  88. data/lib/rails_error_dashboard/services/error_notification_dispatcher.rb +13 -0
  89. data/lib/rails_error_dashboard/services/exception_filter.rb +55 -0
  90. data/lib/rails_error_dashboard/services/git_head_reader.rb +102 -0
  91. data/lib/rails_error_dashboard/services/notification_throttler.rb +172 -28
  92. data/lib/rails_error_dashboard/services/sensitive_data_filter.rb +47 -1
  93. data/lib/rails_error_dashboard/services/storm_protection/circuit_breaker.rb +59 -3
  94. data/lib/rails_error_dashboard/services/storm_protection/count_buffer.rb +64 -6
  95. data/lib/rails_error_dashboard/services/storm_protection/fingerprint_buckets.rb +25 -2
  96. data/lib/rails_error_dashboard/services/storm_protection/gate.rb +32 -5
  97. data/lib/rails_error_dashboard/services/swallowed_exception_tracker.rb +99 -23
  98. data/lib/rails_error_dashboard/services/url_safety.rb +40 -0
  99. data/lib/rails_error_dashboard/services/variable_serializer.rb +125 -10
  100. data/lib/rails_error_dashboard/subscribers/breadcrumb_subscriber.rb +111 -0
  101. data/lib/rails_error_dashboard/subscribers/issue_tracker_subscriber.rb +11 -2
  102. data/lib/rails_error_dashboard/value_objects/error_context.rb +40 -3
  103. data/lib/rails_error_dashboard/version.rb +1 -1
  104. data/lib/rails_error_dashboard.rb +34 -0
  105. data/lib/tasks/error_dashboard.rake +54 -4
  106. metadata +16 -2
@@ -10,8 +10,9 @@ module RailsErrorDashboard
10
10
  # and two concurrent captures could both read count N and both write N+1.
11
11
  #
12
12
  # Search order:
13
+ # 0. wont_fix errors with same hash (any age) → increment, status untouched
13
14
  # 1. Unresolved errors with same hash within 24 hours → increment occurrence count
14
- # 2. Resolved/wont_fix errors with same hash (any age) → reopen and increment
15
+ # 2. Resolved errors with same hash (any age) → reopen and increment
15
16
  # 3. No match → create new error record
16
17
  #
17
18
  # Environment is a MATCH dimension, not part of the hash: the same error in
@@ -42,11 +43,15 @@ module RailsErrorDashboard
42
43
 
43
44
  def call
44
45
  ErrorLog.transaction do
46
+ # Priority 0: a wont_fix row absorbs its recurrences, at any age
47
+ sticky = find_wont_fix
48
+ next increment_existing(sticky) if sticky
49
+
45
50
  # Priority 1: Find unresolved match (existing behavior)
46
51
  existing = find_unresolved
47
52
  next increment_existing(existing) if existing
48
53
 
49
- # Priority 2: Find resolved/wont_fix match → reopen
54
+ # Priority 2: Find resolved match → reopen
50
55
  resolved = find_resolved
51
56
  next reopen_existing(resolved) if resolved
52
57
 
@@ -57,9 +62,26 @@ module RailsErrorDashboard
57
62
 
58
63
  private
59
64
 
65
+ # The three lookups are DISJOINT by status, so which row a recurrence
66
+ # lands on never depends on the order they happen to run in.
67
+
68
+ # "Won't fix" is a decision that the error recurs and will not be acted
69
+ # on, so it has no time window: the row counts its recurrences for as
70
+ # long as it keeps the status. It used to be matched by find_unresolved
71
+ # for 24 hours and then REOPENED by find_resolved -- sticky for a day,
72
+ # after which the triage decision was silently thrown away.
73
+ def find_wont_fix
74
+ with_environment(
75
+ ErrorLog
76
+ .where(error_hash: @error_hash)
77
+ .where(application_id: @attributes[:application_id])
78
+ .where(status: "wont_fix")
79
+ ).lock.order(last_seen_at: :desc).first
80
+ end
81
+
60
82
  def find_unresolved
61
83
  with_environment(
62
- ErrorLog.unresolved
84
+ not_wont_fix(ErrorLog.unresolved)
63
85
  .where(error_hash: @error_hash)
64
86
  .where(application_id: @attributes[:application_id])
65
87
  .where("occurred_at >= ?", 24.hours.ago)
@@ -71,10 +93,16 @@ module RailsErrorDashboard
71
93
  ErrorLog
72
94
  .where(error_hash: @error_hash)
73
95
  .where(application_id: @attributes[:application_id])
74
- .where(status: %w[resolved wont_fix])
96
+ .where(status: "resolved")
75
97
  ).lock.order(last_seen_at: :desc).first
76
98
  end
77
99
 
100
+ # Spelled out rather than where.not(status: "wont_fix"): in SQL that also
101
+ # drops every row whose status is NULL.
102
+ def not_wont_fix(scope)
103
+ scope.where("status IS NULL OR status <> ?", "wont_fix")
104
+ end
105
+
78
106
  # Restrict to this occurrence's environment or a legacy NULL row, exact
79
107
  # first. Literal SQL, no interpolation. A blank environment (column not
80
108
  # migrated yet, or an attribute-less caller) leaves the scope unchanged.
@@ -109,21 +137,78 @@ module RailsErrorDashboard
109
137
  # leaves both the snapshot and its provenance alone -- otherwise the row
110
138
  # would claim a fresh capture time for evidence from an older event,
111
139
  # which is precisely the confusion this exists to remove.
112
- def context_provenance(refreshed)
140
+ def context_provenance(refreshed, error = nil)
113
141
  return {} unless refreshed.any? || refreshed_request_identity?
114
142
  return {} unless ErrorLog.column_names.include?("context_captured_at")
115
143
 
116
144
  provenance = { context_captured_at: @attributes[:occurred_at] || Time.current }
117
145
  if ErrorLog.column_names.include?("context_fidelity")
118
- provenance[:context_fidelity] = @attributes[:_context_fidelity].presence || "full"
146
+ provenance[:context_fidelity] =
147
+ @attributes[:_context_fidelity].presence || snapshot_fidelity(error)
119
148
  end
120
149
  provenance
121
150
  end
122
151
 
152
+ # "full" means every displayed field came from THIS occurrence. When the
153
+ # `||` chain keeps an older value beside a newly refreshed one, the row
154
+ # is showing a mixture of two events, and calling that a fresh full
155
+ # capture is what made the snapshot unreadable: a new request URL sat
156
+ # beside a previous occurrence's user and locals under one timestamp.
157
+ #
158
+ # Keeping the older value is still the right behaviour -- a useful
159
+ # exemplar beats a blank one -- so only the LABEL changes.
160
+ def snapshot_fidelity(error)
161
+ return "full" if error.nil?
162
+
163
+ retains_older_value?(error) ? "partial" : "full"
164
+ end
165
+
166
+ # Every field whose stored value this capture could RETAIN from an
167
+ # earlier occurrence. Defined once, so a field cannot join the displayed
168
+ # snapshot without joining the provenance policy -- which is exactly how
169
+ # local variables came to be shown beside a "full" label and a newer
170
+ # timestamp while belonging to a different event.
171
+ #
172
+ # Request identity plus the context payloads: both are subject to the
173
+ # same `||` chain, and a reader cannot tell them apart on the page.
174
+ PROVENANCE_TRACKED = (REFRESHED_REQUEST_IDENTITY + REFRESHED_CONTEXT).uniq.freeze
175
+
176
+ # True when the row already holds a displayed value that this occurrence
177
+ # did NOT supply, so the `||` chain is about to keep it and the stored
178
+ # snapshot will describe two different events.
179
+ def retains_older_value?(error)
180
+ PROVENANCE_TRACKED.any? do |key|
181
+ next false unless ErrorLog.column_names.include?(key.to_s)
182
+
183
+ @attributes[key].nil? && previous_value(error, key).present?
184
+ end
185
+ end
186
+
187
+ # read_attribute, never public_send: :instance_variables would otherwise
188
+ # resolve to Ruby's own Object#instance_variables if the generated
189
+ # attribute method were ever absent (ignored_columns, load order), and
190
+ # silently compare an Array of symbols against captured context.
191
+ def previous_value(error, key)
192
+ error.read_attribute(key)
193
+ rescue StandardError
194
+ nil
195
+ end
196
+
123
197
  def refreshed_request_identity?
124
198
  REFRESHED_REQUEST_IDENTITY.any? { |key| !@attributes[key].nil? }
125
199
  end
126
200
 
201
+ # Inside the 24-hour matching window find_unresolved uses, so the group
202
+ # this creates can still be found by its own recurrences.
203
+ def clamped_group_time(time)
204
+ return Time.current if time.blank?
205
+
206
+ floor = 24.hours.ago + 1.minute
207
+ time < floor ? floor : time
208
+ rescue StandardError
209
+ Time.current
210
+ end
211
+
127
212
  # A group first seen during a storm has a MINIMAL exemplar: the flush job
128
213
  # could only record the first app frame, because a counted-only event
129
214
  # captures no backtrace. The comment there promises the next occurrence
@@ -177,7 +262,7 @@ module RailsErrorDashboard
177
262
  user_agent: @attributes[:user_agent] || error.user_agent,
178
263
  ip_address: @attributes[:ip_address] || error.ip_address,
179
264
  **refreshed,
180
- **context_provenance(refreshed),
265
+ **context_provenance(refreshed, error),
181
266
  **backtrace_upgrade(error),
182
267
  **environment_adoption(error)
183
268
  )
@@ -197,7 +282,7 @@ module RailsErrorDashboard
197
282
  user_agent: @attributes[:user_agent] || error.user_agent,
198
283
  ip_address: @attributes[:ip_address] || error.ip_address,
199
284
  **(refreshed = latest_context),
200
- **context_provenance(refreshed),
285
+ **context_provenance(refreshed, error),
201
286
  **backtrace_upgrade(error),
202
287
  **environment_adoption(error)
203
288
  }
@@ -217,6 +302,15 @@ module RailsErrorDashboard
217
302
  attrs = @attributes.reject { |key, _| key.to_s.start_with?("_") }
218
303
  attrs = attrs.reverse_merge(resolved: false)
219
304
 
305
+ # The GROUP's occurred_at is clamped into the matching window, even
306
+ # when the EVENT is older. find_unresolved matches on
307
+ # `occurred_at >= 24.hours.ago`, so a backdated report (a mobile client
308
+ # flushing a queue it collected offline) would otherwise create a row
309
+ # that can never be matched again -- every recurrence making yet
310
+ # another group. The event's true time is preserved on its occurrence
311
+ # row, which is what the time-window queries read.
312
+ attrs[:occurred_at] = clamped_group_time(attrs[:occurred_at])
313
+
220
314
  if ErrorLog.column_names.include?("context_captured_at")
221
315
  attrs[:context_captured_at] ||= @attributes[:occurred_at] || Time.current
222
316
  end
@@ -235,9 +329,13 @@ module RailsErrorDashboard
235
329
  ErrorLog.create!(new_record_attributes)
236
330
  end
237
331
  rescue ActiveRecord::RecordNotUnique
238
- # Race condition: another process created the same error
332
+ # Race condition: another process created the same error. Same three
333
+ # lookups, same order, as the first pass.
334
+ retry_sticky = find_wont_fix
335
+ return increment_existing(retry_sticky) if retry_sticky
336
+
239
337
  retry_existing = with_environment(
240
- ErrorLog.unresolved
338
+ not_wont_fix(ErrorLog.unresolved)
241
339
  .where(error_hash: @error_hash)
242
340
  .where(application_id: @attributes[:application_id])
243
341
  .where("occurred_at >= ?", 24.hours.ago)
@@ -257,7 +355,7 @@ module RailsErrorDashboard
257
355
  ErrorLog
258
356
  .where(error_hash: @error_hash)
259
357
  .where(application_id: @attributes[:application_id])
260
- .where(status: %w[resolved wont_fix])
358
+ .where(status: "resolved")
261
359
  ).lock.first
262
360
 
263
361
  if retry_resolved
@@ -28,8 +28,10 @@ module RailsErrorDashboard
28
28
  app_id = current_application_id
29
29
 
30
30
  @counts.each do |key, count|
31
+ # Path and user agent are attacker-supplied, so invalid bytes are the
32
+ # expected case here. Scrub the whole key before it is split.
31
33
  rule, match_type, discriminator, path, http_method, user_agent =
32
- Services::RackAttackTracker.parse_key(key)
34
+ Services::RackAttackTracker.parse_key(Services::EncodingSanitizer.scrub(key.to_s))
33
35
 
34
36
  next if rule.blank? || match_type.blank?
35
37
 
@@ -16,13 +16,26 @@ module RailsErrorDashboard
16
16
  # Counts are exact. Notifications are NOT dispatched from here — during a
17
17
  # storm they're suppressed by design; the storm notification covers it.
18
18
  class FlushStormCounts
19
+ # A bucket write failed in a way that may succeed on retry. Raised so the
20
+ # per-entry rescue in #call classifies it exactly as it classifies a
21
+ # transient COUNT failure: roll the batch back, leave the ledger
22
+ # unclaimed, let the job retry the whole batch intact.
23
+ class EventCountWriteFailed < StandardError; end
24
+
19
25
  def self.call(entries:, overflow: 0, episode: nil, batch_id: nil)
20
26
  new(entries: entries, overflow: overflow, episode: episode, batch_id: batch_id).call
21
27
  end
22
28
 
23
29
  def initialize(entries:, overflow: 0, episode: nil, batch_id: nil)
24
- @entries = Array(entries)
30
+ # The gate scrubs what it buffers, so this is normally a no-op scan. It
31
+ # covers entries buffered by an older release and direct callers: the
32
+ # exemplar becomes an ErrorLog row, and its message is matched by regex.
33
+ @entries = Array(Services::EncodingSanitizer.scrub_deep(entries))
25
34
  @overflow = overflow.to_i
35
+ # Set when a bucket write was permanently unavailable. The counts are
36
+ # still correct; only their placement in TIME is missing, and a caller
37
+ # reading a time window deserves to know that.
38
+ @buckets_incomplete = false
26
39
  @episode = episode
27
40
  @batch_id = batch_id
28
41
  end
@@ -31,6 +44,7 @@ module RailsErrorDashboard
31
44
  application = resolve_application
32
45
  counted = 0
33
46
  failed = 0
47
+ aborted = false
34
48
 
35
49
  # Counts are applied additively (occurrence_count + N), which is not
36
50
  # idempotent: delivering the same snapshot twice counted it twice. The
@@ -52,6 +66,20 @@ module RailsErrorDashboard
52
66
  @entries.each do |entry|
53
67
  entry = entry.with_indifferent_access if entry.respond_to?(:with_indifferent_access)
54
68
  counted += reconcile_entry(entry, application)
69
+ rescue EventCountWriteFailed, *Commands::LogError::RETRYABLE_STORE_ERRORS => e
70
+ # A transient store failure is NOT a bad entry. Claiming the batch
71
+ # here would commit the ledger row and strand every entry not yet
72
+ # applied: the retry is then suppressed as a replay and those events
73
+ # are lost for good. Re-raise so the whole transaction rolls back --
74
+ # nothing was committed, so nothing can double -- and let the job
75
+ # retry the batch intact. The generic rescue below still keeps a
76
+ # permanently malformed entry from poisoning its batch.
77
+ failed += 1
78
+ aborted = true
79
+ RailsErrorDashboard::Logger.error(
80
+ "[RailsErrorDashboard] Storm batch aborted by a transient store failure: #{e.class} - #{e.message}"
81
+ )
82
+ raise
55
83
  rescue => e
56
84
  # A corrupt (non-Hash) entry must not abort the whole batch — and the
57
85
  # log line itself must not assume `entry` is subscriptable (an Integer
@@ -67,26 +95,56 @@ module RailsErrorDashboard
67
95
  # roll the claim back and let the job retry the whole batch.
68
96
  raise ActiveRecord::Rollback if failed.positive? && counted.zero?
69
97
 
98
+ # INSIDE the transaction, deliberately.
99
+ #
100
+ # The counts and the record that their timing is unreliable have to
101
+ # land together or not at all. Writing this after the commit (as the
102
+ # storm-episode marker did) meant a transient failure lost the
103
+ # marker while the ledger had already recorded the batch as applied
104
+ # -- the replay was then suppressed and the gap was never recorded,
105
+ # so the dashboard reported completeness it could not vouch for.
106
+ #
107
+ # If this write fails, the whole batch rolls back and stays
108
+ # replayable. Counts whose unreliability we cannot record are worth
109
+ # retrying, not committing silently.
110
+ record_timing_gap!(counted) if @buckets_incomplete && counted.positive?
111
+
70
112
  finalize_batch!(ledger, counted)
71
113
  end
72
114
 
73
115
  # Every entry failed and none was written. Reporting success with
74
116
  # reconciled: 0 made a total loss indistinguishable from an empty
75
117
  # batch, so the job acknowledged counts that never reached the
76
- # database. Partial success stays successful: the entries that were
77
- # written are written, and replaying the batch would double them.
118
+ # database.
119
+ #
120
+ # Reaching here means every failure was PERMANENT -- a transient store
121
+ # failure re-raises above and rolls the whole batch back. Partial
122
+ # success over permanent failures stays successful: the entries that
123
+ # were written are written, replaying would double them, and retrying a
124
+ # corrupt payload only loops forever.
78
125
  if failed.positive? && counted.zero?
79
- return { success: false, reconciled: 0, failed: failed, overflow: @overflow,
126
+ return { success: false, retryable: false, reconciled: 0, failed: failed, overflow: @overflow,
127
+ buckets_incomplete: @buckets_incomplete,
80
128
  error: "all #{failed} entries failed to reconcile" }
81
129
  end
82
130
 
83
131
  upsert_storm_event(counted)
84
- { success: true, reconciled: counted, failed: failed, overflow: @overflow }
132
+ result = { success: true, reconciled: counted, failed: failed, overflow: @overflow }
133
+ result[:buckets_incomplete] = true if @buckets_incomplete
134
+ result
85
135
  rescue => e
86
136
  RailsErrorDashboard::Logger.error(
87
137
  "[RailsErrorDashboard] FlushStormCounts failed: #{e.class} - #{e.message}"
88
138
  )
89
- { success: false, error: "#{e.class}: #{e.message}" }
139
+ # retryable: true says "the batch is intact, replay it" -- nothing was
140
+ # committed, so the job can retry without doubling. A permanent failure
141
+ # carries no such promise.
142
+ retryable = Commands::LogError::RETRYABLE_STORE_ERRORS.any? { |klass| e.is_a?(klass) }
143
+ result = { success: false, retryable: retryable, error: "#{e.class}: #{e.message}" }
144
+ # reconciled: 0 because the transaction rolled back -- whatever this
145
+ # batch had counted in memory never reached the database.
146
+ result.merge!(reconciled: 0, failed: failed, overflow: @overflow) if aborted
147
+ result
90
148
  end
91
149
 
92
150
  private
@@ -145,7 +203,68 @@ module RailsErrorDashboard
145
203
  end
146
204
 
147
205
 
206
+ # Reconcile one buffered entry and, on every path that adds counts, give
207
+ # those shed events a TIME BUCKET as well as a total.
208
+ #
209
+ # occurrence_count alone is a lifetime counter: it says how many events
210
+ # there were but not when, so no window query can place them. Ordinary
211
+ # captures carry their own ErrorOccurrence row; shed events write none by
212
+ # design, which is what made them invisible to "errors today". The bucket
213
+ # written here is what Queries::EventVolume adds to the occurrence rows.
148
214
  def reconcile_entry(entry, application)
215
+ error_log_id = nil
216
+ count = reconcile_entry_count(entry, application) { |id| error_log_id = id }
217
+ write_event_buckets(entry, error_log_id, count) if count.positive? && error_log_id
218
+ count
219
+ end
220
+
221
+ # Write one row per BUCKET the producer recorded, not one row for the
222
+ # whole entry.
223
+ #
224
+ # The buffer tallies events per 15-minute bucket precisely so this does
225
+ # not have to guess: assigning the entry's whole total to last_seen_at's
226
+ # bucket put an event from 23:59:59 and one from 00:00:01 on the same
227
+ # day. A payload from an older release carries no buckets, so it falls
228
+ # back to the old behaviour rather than losing the count.
229
+ def write_event_buckets(entry, error_log_id, count)
230
+ buckets = entry["buckets"]
231
+ buckets = nil unless buckets.is_a?(Hash) && buckets.any?
232
+
233
+ pairs =
234
+ if buckets
235
+ buckets.map { |at, n| [ Time.zone.at(at.to_i), n.to_i ] }
236
+ else
237
+ [ [ parse_time(entry["last_seen_at"]) || Time.current, count ] ]
238
+ end
239
+
240
+ pairs.each do |bucket_at, n|
241
+ next unless n.positive?
242
+
243
+ # Three outcomes, three responses.
244
+ #
245
+ # :written -- done.
246
+ # raises -- TRANSIENT. Handled by the caller's rescue, which
247
+ # aborts and replays the batch intact. Swallowing it
248
+ # finalized the ledger with the bucket missing, so
249
+ # the replay was suppressed as already-applied and
250
+ # the time window lost those events permanently
251
+ # while the lifetime count stayed correct.
252
+ # :unavailable -- PERMANENT. Degrade: the rollup is simply not
253
+ # usable on this host (never migrated, adapter
254
+ # refuses the statement). Raising here rolled back
255
+ # the surrounding transaction and destroyed the
256
+ # authoritative lifetime count along with it --
257
+ # turning a missing time bucket into a lost count,
258
+ # which is strictly worse. Record it instead, so the
259
+ # result can say the timing evidence is incomplete.
260
+ case EventCount.accumulate(error_log_id: error_log_id, bucket_at: bucket_at, count: n)
261
+ when :written then next
262
+ else @buckets_incomplete = true
263
+ end
264
+ end
265
+ end
266
+
267
+ def reconcile_entry_count(entry, application)
149
268
  count = entry["count"].to_i
150
269
  return 0 if count <= 0
151
270
 
@@ -165,7 +284,12 @@ module RailsErrorDashboard
165
284
  # long-running unresolved error legitimately owns several unresolved
166
285
  # rows. An update_all across the whole hash would add N to every one
167
286
  # of them — seven real events becoming twelve counted occurrences.
168
- target = unresolved_target(error_hash, application, env)
287
+ #
288
+ # Priority 0, tried first: a wont_fix row absorbs its recurrences at any
289
+ # age and keeps its status (FindOrIncrementError#find_wont_fix). Same
290
+ # atomic increment, so it shares the branch below.
291
+ target = wont_fix_target(error_hash, application, env) ||
292
+ unresolved_target(error_hash, application, env)
169
293
  if target
170
294
  if env && target.environment.blank?
171
295
  ErrorLog.where(id: target.id).update_all([
@@ -177,30 +301,56 @@ module RailsErrorDashboard
177
301
  "occurrence_count = occurrence_count + ?, last_seen_at = ?", count, last_seen
178
302
  ])
179
303
  end
304
+ yield target.id if block_given?
180
305
  return count
181
306
  end
182
307
 
183
- # Priority 2: resolved/wont_fix match — reopen, mirroring
308
+ # Priority 2: resolved match — reopen, mirroring
184
309
  # FindOrIncrementError so storm recurrences don't stay buried
185
310
  resolved_scope = ErrorLog
186
311
  .where(error_hash: error_hash, application_id: application.id)
187
- .where(status: %w[resolved wont_fix])
312
+ .where(status: "resolved")
188
313
  if env
189
314
  resolved_scope = resolved_scope.where(environment: [ env, nil ])
190
315
  .order(Arel.sql("CASE WHEN environment IS NULL THEN 1 ELSE 0 END"))
191
316
  end
192
- resolved = resolved_scope.order(last_seen_at: :desc).first
317
+ # .lock (SELECT ... FOR UPDATE) held to commit by the transaction opened
318
+ # in #call, exactly as FindOrIncrementError does for the same reopen.
319
+ # Without it this branch read occurrence_count into Ruby and wrote an
320
+ # ABSOLUTE value back, so two concurrent batches both read N and both
321
+ # wrote N+count -- one batch's events vanished while both reported
322
+ # success. The unresolved branch above is safe because it increments in
323
+ # SQL; this branch cannot use update_all because reopening is a state
324
+ # transition the dashboard must see, and update_all skips the
325
+ # after_update_commit broadcast.
326
+ resolved = resolved_scope.lock.order(last_seen_at: :desc).first
193
327
  if resolved
328
+ # The count is incremented in SQL, never read into Ruby and written
329
+ # back. This branch used to compute `resolved.occurrence_count + count`
330
+ # and write that ABSOLUTE value, so two concurrent batches both read N
331
+ # and both wrote N+count -- one batch's events vanished while both
332
+ # reported success. The .lock above serializes the pair on
333
+ # PostgreSQL/MySQL; the atomic increment below conserves the count on
334
+ # every adapter, including SQLite where FOR UPDATE is a no-op.
335
+ ErrorLog.where(id: resolved.id).update_all([
336
+ "occurrence_count = occurrence_count + ?", count
337
+ ])
338
+
339
+ # The reopen is a state transition the dashboard must see, so it stays
340
+ # an update! -- update_all would skip the after_update_commit
341
+ # broadcast. Reload first so this write does not clobber the increment
342
+ # just made with a stale in-memory occurrence_count.
343
+ resolved.reload
194
344
  attrs = {
195
345
  resolved: false,
196
346
  status: "new",
197
347
  resolved_at: nil,
198
- occurrence_count: resolved.occurrence_count + count,
199
348
  last_seen_at: last_seen
200
349
  }
201
350
  attrs[:reopened_at] = Time.current if ErrorLog.column_names.include?("reopened_at")
202
351
  attrs[:environment] = env if env && resolved.environment.blank?
203
352
  resolved.update!(attrs)
353
+ yield resolved.id if block_given?
204
354
  return count
205
355
  end
206
356
 
@@ -241,14 +391,67 @@ module RailsErrorDashboard
241
391
  if ErrorLog.column_names.include?("context_captured_at")
242
392
  create_attrs[:context_captured_at] = create_attrs[:occurred_at]
243
393
  end
244
- ErrorLog.create!(**ErrorLog.clamp_string_attributes(Services::SensitiveDataFilter.filter_attributes(create_attrs)))
394
+ begin
395
+ # requires_new opens a SAVEPOINT: on PostgreSQL a failed INSERT aborts
396
+ # its transaction, and every later statement -- including the recovery
397
+ # lookup below -- fails with InFailedSqlTransaction. The savepoint
398
+ # confines the damage to this INSERT so the batch can continue.
399
+ ErrorLog.transaction(requires_new: true) do
400
+ created = ErrorLog.create!(**ErrorLog.clamp_string_attributes(Services::SensitiveDataFilter.filter_attributes(create_attrs)))
401
+ yield created.id if block_given?
402
+ end
403
+ rescue ActiveRecord::RecordNotUnique
404
+ # Another flush created this group between our lookups and this
405
+ # INSERT -- the group-identity index objected. Two concurrent batches
406
+ # for the same fingerprint is the NORMAL storm shape (every process
407
+ # flushes its own batch), so this must not fail the entry: the counts
408
+ # exist nowhere but in this payload. Re-run the same lookups and add
409
+ # to the row that now exists, exactly as FindOrIncrementError does.
410
+ #
411
+ # Nested in its own transaction because the failed INSERT poisons the
412
+ # surrounding one on PostgreSQL.
413
+ raise unless (target = existing_target(error_hash, application, env))
414
+
415
+ ErrorLog.where(id: target.id).update_all([
416
+ "occurrence_count = occurrence_count + ?, last_seen_at = ?", count, last_seen
417
+ ])
418
+ # Yield on the RECOVERY path too: these counts are as real as the ones
419
+ # the winning INSERT wrote, so they need a time bucket as well, or a
420
+ # raced create silently loses its volume from every window query.
421
+ yield target.id if block_given?
422
+ end
245
423
  count
246
424
  end
247
425
 
426
+ # The row a retried INSERT should add to: any row holding this group
427
+ # identity, whatever its status. Deliberately wider than the
428
+ # priority-ordered lookups above -- the index has already proved a row
429
+ # with this identity exists, so refusing to match a resolved or wont_fix
430
+ # one would drop the counts instead.
431
+ def existing_target(error_hash, application, env)
432
+ scope = ErrorLog.where(error_hash: error_hash, application_id: application.id)
433
+ scope = scope.where(environment: [ env, nil ]) if env
434
+ scope.order(last_seen_at: :desc).select(:id).first
435
+ end
248
436
 
249
437
  # The unresolved row the full capture path would increment right now.
438
+ # No time window: "won't fix" holds for as long as the row keeps the status.
439
+ def wont_fix_target(error_hash, application, env)
440
+ scope = ErrorLog
441
+ .where(error_hash: error_hash, application_id: application.id)
442
+ .where(status: "wont_fix")
443
+ if env
444
+ scope = scope.where(environment: [ env, nil ])
445
+ .order(Arel.sql("CASE WHEN environment IS NULL THEN 1 ELSE 0 END"))
446
+ end
447
+ scope.order(last_seen_at: :desc).select(:id, :environment).first
448
+ end
449
+
250
450
  def unresolved_target(error_hash, application, env)
451
+ # Disjoint from wont_fix_target. Not where.not(...): that would also
452
+ # drop rows whose status is NULL.
251
453
  scope = ErrorLog.unresolved
454
+ .where("status IS NULL OR status <> ?", "wont_fix")
252
455
  .where(error_hash: error_hash, application_id: application.id)
253
456
  .where("occurred_at >= ?", 24.hours.ago)
254
457
  if env
@@ -303,6 +506,36 @@ module RailsErrorDashboard
303
506
  Application.find_or_create_by_name(app_name)
304
507
  end
305
508
 
509
+ # Record the interval whose per-event timing was lost.
510
+ #
511
+ # Keyed by the interval, NOT by a storm episode: the gate can shed events
512
+ # with its breaker closed and pass episode: nil, and a marker on the
513
+ # episode vanished in exactly that case. The events are just as
514
+ # untimed whether or not an episode object happens to exist.
515
+ #
516
+ # Bounds come from the entries themselves, so the gap describes when the
517
+ # events actually happened rather than when the worker got to them.
518
+ def record_timing_gap!(counted)
519
+ return unless EventTimingGap.table_exists?
520
+
521
+ times = @entries.filter_map do |entry|
522
+ entry = entry.with_indifferent_access if entry.respond_to?(:with_indifferent_access)
523
+ parse_time(entry["last_seen_at"]) || parse_time(entry["first_seen_at"])
524
+ end
525
+ first_seen = @entries.filter_map do |entry|
526
+ entry = entry.with_indifferent_access if entry.respond_to?(:with_indifferent_access)
527
+ parse_time(entry["first_seen_at"])
528
+ end
529
+
530
+ now = Time.current
531
+ EventTimingGap.create!(
532
+ application_id: resolve_application&.id,
533
+ covered_from: (first_seen + times).min || now,
534
+ covered_until: times.max || now,
535
+ events_affected: counted
536
+ )
537
+ end
538
+
306
539
  def upsert_storm_event(counted)
307
540
  return unless @episode.is_a?(Hash)
308
541
  return unless StormEvent.table_exists?
@@ -323,6 +556,13 @@ module RailsErrorDashboard
323
556
  event.fingerprints_affected = [ event.fingerprints_affected.to_i, @entries.size ].max
324
557
  event.peak_rate_per_minute = [ event.peak_rate_per_minute.to_i, @episode["peak_rate_per_minute"].to_i ].max
325
558
  event.reached_open ||= @episode["reached_open"] == true
559
+ # Sticky, like reached_open: once an episode has lost bucket timing it
560
+ # has lost it, and a later flush that happens to succeed does not make
561
+ # the earlier gap reappear. Guarded on the column so a host that has
562
+ # not run the migration yet keeps flushing normally.
563
+ if @buckets_incomplete && event.respond_to?(:buckets_incomplete)
564
+ event.buckets_incomplete = true
565
+ end
326
566
  event.top_fingerprints = top_fingerprints_json(event)
327
567
  event.ended_at = parse_time(@episode["ended_at"]) if @episode["ended_at"]
328
568
  event.save!
@@ -25,8 +25,11 @@ module RailsErrorDashboard
25
25
  app_id = current_application_id
26
26
 
27
27
  # Process raise counts
28
+ # Keys are scrubbed before they are split: a class name or path with an
29
+ # invalid byte makes split/blank? raise, and the outer rescue would then
30
+ # drop every remaining count in the batch along with it.
28
31
  @raise_counts.each do |key, count|
29
- class_name, location = key.split("|", 2)
32
+ class_name, location = Services::EncodingSanitizer.scrub(key.to_s).split("|", 2)
30
33
  next if class_name.blank? || location.blank?
31
34
 
32
35
  upsert_raise(class_name, location, period, app_id, count)
@@ -34,7 +37,7 @@ module RailsErrorDashboard
34
37
 
35
38
  # Process rescue counts
36
39
  @rescue_counts.each do |key, count|
37
- class_name, locations = key.split("|", 2)
40
+ class_name, locations = Services::EncodingSanitizer.scrub(key.to_s).split("|", 2)
38
41
  next if class_name.blank? || locations.blank?
39
42
 
40
43
  raise_loc, rescue_loc = locations.split("->", 2)
@@ -32,6 +32,7 @@ module RailsErrorDashboard
32
32
 
33
33
  def call
34
34
  return { success: false, error: red_t("red.commands.issue.url_required") } if @issue_url.blank?
35
+ return { success: false, error: red_t("red.commands.issue.url_invalid") } unless Services::UrlSafety.http_url?(@issue_url)
35
36
 
36
37
  error = ErrorLog.find(@error_id)
37
38
  parsed = parse_issue_url(@issue_url)