rails_error_dashboard 0.12.1 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. checksums.yaml +4 -4
  2. data/app/controllers/rails_error_dashboard/application_controller.rb +92 -11
  3. data/app/controllers/rails_error_dashboard/errors_controller.rb +111 -34
  4. data/app/controllers/rails_error_dashboard/webhooks_controller.rb +3 -0
  5. data/app/helpers/rails_error_dashboard/application_helper.rb +46 -1
  6. data/app/helpers/rails_error_dashboard/backtrace_helper.rb +8 -0
  7. data/app/jobs/rails_error_dashboard/async_error_logging_job.rb +10 -0
  8. data/app/jobs/rails_error_dashboard/concerns/plain_channel_message.rb +71 -0
  9. data/app/jobs/rails_error_dashboard/notification_burst_summary_job.rb +55 -0
  10. data/app/jobs/rails_error_dashboard/retention_cleanup_job.rb +122 -1
  11. data/app/jobs/rails_error_dashboard/storm_flush_job.rb +7 -4
  12. data/app/jobs/rails_error_dashboard/storm_notification_job.rb +8 -44
  13. data/app/models/rails_error_dashboard/error_baseline.rb +12 -9
  14. data/app/models/rails_error_dashboard/error_comment.rb +0 -5
  15. data/app/models/rails_error_dashboard/error_log.rb +19 -3
  16. data/app/models/rails_error_dashboard/error_logs_record.rb +34 -0
  17. data/app/models/rails_error_dashboard/error_occurrence.rb +9 -1
  18. data/app/models/rails_error_dashboard/event_count.rb +132 -0
  19. data/app/models/rails_error_dashboard/event_timing_gap.rb +55 -0
  20. data/app/views/layouts/rails_error_dashboard.html.erb +58 -5
  21. data/app/views/rails_error_dashboard/errors/_discussion.html.erb +18 -11
  22. data/app/views/rails_error_dashboard/errors/_issue_section.html.erb +22 -8
  23. data/app/views/rails_error_dashboard/errors/_request_context.html.erb +2 -0
  24. data/app/views/rails_error_dashboard/errors/_sidebar_metadata.html.erb +15 -6
  25. data/app/views/rails_error_dashboard/errors/analytics.html.erb +8 -8
  26. data/app/views/rails_error_dashboard/errors/diagnostic_dumps.html.erb +1 -1
  27. data/app/views/rails_error_dashboard/errors/index.html.erb +6 -1
  28. data/app/views/rails_error_dashboard/errors/overview.html.erb +12 -0
  29. data/app/views/rails_error_dashboard/errors/platform_comparison.html.erb +7 -7
  30. data/app/views/rails_error_dashboard/errors/releases.html.erb +2 -2
  31. data/app/views/rails_error_dashboard/errors/settings.html.erb +2 -0
  32. data/app/views/rails_error_dashboard/errors/show.html.erb +3 -3
  33. data/config/locales/de.yml +30 -0
  34. data/config/locales/en.yml +37 -0
  35. data/config/locales/es.yml +30 -0
  36. data/config/locales/fr.yml +30 -0
  37. data/config/locales/it.yml +30 -0
  38. data/config/locales/ja.yml +30 -0
  39. data/config/locales/pl.yml +30 -0
  40. data/config/locales/pt-BR.yml +30 -0
  41. data/config/locales/ru.yml +30 -0
  42. data/config/locales/uk.yml +30 -0
  43. data/config/locales/zh-CN.yml +30 -0
  44. data/db/migrate/20260917000001_add_last_notified_at_to_error_logs.rb +40 -0
  45. data/db/migrate/20260919000001_create_event_counts.rb +71 -0
  46. data/db/migrate/20260920000001_add_buckets_incomplete_to_storm_events.rb +25 -0
  47. data/db/migrate/20260920000002_create_event_timing_gaps.rb +55 -0
  48. data/lib/generators/rails_error_dashboard/install/templates/initializer.rb +12 -1
  49. data/lib/rails_error_dashboard/commands/assign_error.rb +12 -2
  50. data/lib/rails_error_dashboard/commands/backfill_environments.rb +2 -0
  51. data/lib/rails_error_dashboard/commands/backfill_resolved_at.rb +42 -0
  52. data/lib/rails_error_dashboard/commands/batch_delete_errors.rb +1 -0
  53. data/lib/rails_error_dashboard/commands/batch_mute_errors.rb +2 -0
  54. data/lib/rails_error_dashboard/commands/batch_resolve_errors.rb +2 -0
  55. data/lib/rails_error_dashboard/commands/batch_unmute_errors.rb +2 -0
  56. data/lib/rails_error_dashboard/commands/find_or_increment_error.rb +109 -11
  57. data/lib/rails_error_dashboard/commands/flush_rack_attack_events.rb +3 -1
  58. data/lib/rails_error_dashboard/commands/flush_storm_counts.rb +252 -12
  59. data/lib/rails_error_dashboard/commands/flush_swallowed_exceptions.rb +5 -2
  60. data/lib/rails_error_dashboard/commands/link_existing_issue.rb +1 -0
  61. data/lib/rails_error_dashboard/commands/log_error.rb +291 -40
  62. data/lib/rails_error_dashboard/commands/mute_error.rb +1 -0
  63. data/lib/rails_error_dashboard/commands/resolve_error.rb +2 -0
  64. data/lib/rails_error_dashboard/commands/scrub_invalid_encoding.rb +104 -0
  65. data/lib/rails_error_dashboard/commands/snooze_error.rb +34 -10
  66. data/lib/rails_error_dashboard/commands/unmute_error.rb +1 -0
  67. data/lib/rails_error_dashboard/commands/update_error_priority.rb +23 -2
  68. data/lib/rails_error_dashboard/commands/update_error_status.rb +33 -6
  69. data/lib/rails_error_dashboard/configuration.rb +39 -1
  70. data/lib/rails_error_dashboard/engine.rb +28 -0
  71. data/lib/rails_error_dashboard/manual_error_reporter.rb +16 -5
  72. data/lib/rails_error_dashboard/queries/analytics_stats.rb +89 -30
  73. data/lib/rails_error_dashboard/queries/baseline_stats.rb +107 -0
  74. data/lib/rails_error_dashboard/queries/dashboard_stats.rb +216 -74
  75. data/lib/rails_error_dashboard/queries/error_correlation.rb +6 -3
  76. data/lib/rails_error_dashboard/queries/errors_list.rb +15 -2
  77. data/lib/rails_error_dashboard/queries/event_volume.rb +503 -0
  78. data/lib/rails_error_dashboard/queries/similar_errors.rb +1 -1
  79. data/lib/rails_error_dashboard/services/analytics_cache_manager.rb +43 -18
  80. data/lib/rails_error_dashboard/services/backtrace_processor.rb +3 -1
  81. data/lib/rails_error_dashboard/services/breadcrumb_collector.rb +23 -0
  82. data/lib/rails_error_dashboard/services/cause_chain_extractor.rb +3 -1
  83. data/lib/rails_error_dashboard/services/codeberg_issue_client.rb +13 -2
  84. data/lib/rails_error_dashboard/services/diagnostic_dump_generator.rb +5 -3
  85. data/lib/rails_error_dashboard/services/encoding_sanitizer.rb +80 -0
  86. data/lib/rails_error_dashboard/services/error_broadcaster.rb +180 -32
  87. data/lib/rails_error_dashboard/services/error_hash_generator.rb +9 -3
  88. data/lib/rails_error_dashboard/services/error_notification_dispatcher.rb +13 -0
  89. data/lib/rails_error_dashboard/services/exception_filter.rb +55 -0
  90. data/lib/rails_error_dashboard/services/git_head_reader.rb +102 -0
  91. data/lib/rails_error_dashboard/services/notification_throttler.rb +172 -28
  92. data/lib/rails_error_dashboard/services/sensitive_data_filter.rb +47 -1
  93. data/lib/rails_error_dashboard/services/storm_protection/circuit_breaker.rb +59 -3
  94. data/lib/rails_error_dashboard/services/storm_protection/count_buffer.rb +64 -6
  95. data/lib/rails_error_dashboard/services/storm_protection/fingerprint_buckets.rb +25 -2
  96. data/lib/rails_error_dashboard/services/storm_protection/gate.rb +32 -5
  97. data/lib/rails_error_dashboard/services/swallowed_exception_tracker.rb +99 -23
  98. data/lib/rails_error_dashboard/services/url_safety.rb +40 -0
  99. data/lib/rails_error_dashboard/services/variable_serializer.rb +125 -10
  100. data/lib/rails_error_dashboard/subscribers/breadcrumb_subscriber.rb +111 -0
  101. data/lib/rails_error_dashboard/subscribers/issue_tracker_subscriber.rb +11 -2
  102. data/lib/rails_error_dashboard/value_objects/error_context.rb +40 -3
  103. data/lib/rails_error_dashboard/version.rb +1 -1
  104. data/lib/rails_error_dashboard.rb +34 -0
  105. data/lib/tasks/error_dashboard.rake +54 -4
  106. metadata +16 -2
@@ -0,0 +1,55 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RailsErrorDashboard
4
+ # Sends the ONE message that replaces the rest of a burst of new-error
5
+ # notifications.
6
+ #
7
+ # A bad deploy can produce hundreds of DISTINCT new errors. Each is a first
8
+ # occurrence, so the per-error cooldown never applies; without a cap every
9
+ # one of them pages. NotificationThrottler.burst_decision lets the first
10
+ # config.notification_burst_limit through per window and asks for this job
11
+ # exactly once, when the limit is first exceeded.
12
+ #
13
+ # The message states the limit and the window rather than a count of what
14
+ # was suppressed: it is sent when suppression STARTS, and a count at that
15
+ # moment would always be one. Every error is still stored and counted; only
16
+ # the notifications are held back.
17
+ class NotificationBurstSummaryJob < ApplicationJob
18
+ include Concerns::PlainChannelMessage
19
+
20
+ queue_as :default
21
+
22
+ # @param limit [Integer] config.notification_burst_limit when the cap engaged
23
+ # @param window_seconds [Integer] config.notification_burst_window_seconds
24
+ # @param locale [String, nil] resolved at enqueue time
25
+ def perform(limit:, window_seconds:, locale: nil)
26
+ config = RailsErrorDashboard.configuration
27
+
28
+ deliver_plain_message(build_message(limit, window_seconds, config, job_locale(locale)), {
29
+ event: "new_error_notifications_suppressed",
30
+ limit: limit,
31
+ window_seconds: window_seconds,
32
+ application: app_name(config)
33
+ })
34
+ rescue => e
35
+ Rails.logger.error("[RailsErrorDashboard] NotificationBurstSummaryJob failed: #{e.class} - #{e.message}")
36
+ end
37
+
38
+ private
39
+
40
+ # Whole sentences joined, never fragments: see StormNotificationJob.
41
+ # ":warning:" is Slack/Discord emoji shortcode, not text.
42
+ def build_message(limit, window_seconds, config, locale)
43
+ dashboard = (config.dashboard_base_url || "").chomp("/")
44
+ link = if dashboard.present?
45
+ " " + t("red.notifications.burst.dashboard", locale, url: "#{dashboard}/errors")
46
+ else
47
+ ""
48
+ end
49
+
50
+ ":warning: " \
51
+ "#{t("red.notifications.burst.suppressed", locale, application: app_name(config), limit: limit, window: window_seconds)} " \
52
+ "#{t("red.notifications.burst.still_recorded", locale)}#{link}"
53
+ end
54
+ end
55
+ end
@@ -14,6 +14,25 @@ module RailsErrorDashboard
14
14
  class RetentionCleanupJob < ApplicationJob
15
15
  queue_as :default
16
16
 
17
+ # The errors retention applies to: "not seen for retention_days", not
18
+ # "first seen retention_days ago". occurred_at is stamped when a group is
19
+ # created and never moves, so expiring by it deleted errors that were still
20
+ # happening today -- with their comments and triage history -- the day
21
+ # they turned N days old.
22
+ #
23
+ # Equivalent to COALESCE(last_seen_at, occurred_at) < cutoff (the NULL arm
24
+ # covers rows from before last_seen_at existed), but written so that each
25
+ # arm can use its own index; wrapping the columns in COALESCE would force a
26
+ # full scan of the largest table on every run.
27
+ #
28
+ # Public because the rake task previews the same selection before it asks
29
+ # for confirmation.
30
+ def self.expired_scope(cutoff)
31
+ ErrorLog.where(
32
+ "last_seen_at < :cutoff OR (last_seen_at IS NULL AND occurred_at < :cutoff)", cutoff: cutoff
33
+ )
34
+ end
35
+
17
36
  # @return [Integer] number of errors deleted
18
37
  def perform
19
38
  retention_days = RailsErrorDashboard.configuration.retention_days
@@ -26,8 +45,11 @@ module RailsErrorDashboard
26
45
  # whenever no error logs happen to be expired.
27
46
  cleanup_rack_attack_events(cutoff)
28
47
  cleanup_storm_flush_batches(cutoff)
48
+ cleanup_diagnostic_dumps(cutoff)
49
+ cleanup_swallowed_exceptions(cutoff)
50
+ cleanup_event_timing_gaps(cutoff)
29
51
 
30
- expired_scope = ErrorLog.where("occurred_at < ?", cutoff)
52
+ expired_scope = self.class.expired_scope(cutoff)
31
53
  return 0 if expired_scope.none?
32
54
 
33
55
  deleted_count = 0
@@ -37,6 +59,12 @@ module RailsErrorDashboard
37
59
 
38
60
  # Batch delete dependent records (occurrences, comments, cascade patterns)
39
61
  ErrorOccurrence.where(error_log_id: expired_ids_scope).in_batches(of: 1000).delete_all
62
+ # Hour buckets for storm-shed events. Listed explicitly because this job
63
+ # deletes with delete_all, which does not fire the has_many :dependent
64
+ # callback on ErrorLog -- without this line the buckets outlive the group
65
+ # they describe, forever. The migration's own comment promised this
66
+ # pruning before the code existed.
67
+ EventCount.where(error_log_id: expired_ids_scope).in_batches(of: 1000).delete_all if EventCount.table_exists?
40
68
  ErrorComment.where(error_log_id: expired_ids_scope).in_batches(of: 1000).delete_all
41
69
  CascadePattern.where(parent_error_id: expired_ids_scope)
42
70
  .or(CascadePattern.where(child_error_id: expired_ids_scope))
@@ -48,6 +76,9 @@ module RailsErrorDashboard
48
76
  deleted_count += batch_size
49
77
  end
50
78
 
79
+ # delete_all skips callbacks, and the stat cards are cached.
80
+ Services::AnalyticsCacheManager.clear if deleted_count > 0
81
+
51
82
  if deleted_count > 0
52
83
  RailsErrorDashboard::Logger.info(
53
84
  "[RailsErrorDashboard] Retention cleanup: deleted #{deleted_count} errors older than #{retention_days} days"
@@ -86,6 +117,96 @@ module RailsErrorDashboard
86
117
  )
87
118
  end
88
119
 
120
+ # Diagnostic dumps and swallowed-exception aggregates are written
121
+ # independently of error logs and nothing else ever deletes them, so they
122
+ # grew without bound. Not gated on their feature flags: rows written while
123
+ # a feature was on must still expire after it is switched off. Own rescue
124
+ # each, like the cleanups around them.
125
+ def cleanup_diagnostic_dumps(cutoff)
126
+ return unless DiagnosticDump.table_exists?
127
+
128
+ deleted = 0
129
+ DiagnosticDump.where("captured_at < ?", cutoff).in_batches(of: 1000) do |batch|
130
+ deleted += batch.delete_all
131
+ end
132
+
133
+ if deleted > 0
134
+ RailsErrorDashboard::Logger.info(
135
+ "[RailsErrorDashboard] Retention cleanup: deleted #{deleted} diagnostic dumps"
136
+ )
137
+ end
138
+ rescue => e
139
+ RailsErrorDashboard::Logger.debug(
140
+ "[RailsErrorDashboard] Diagnostic dump retention cleanup failed: #{e.class} - #{e.message}"
141
+ )
142
+ end
143
+
144
+ def cleanup_swallowed_exceptions(cutoff)
145
+ return unless SwallowedException.table_exists?
146
+
147
+ deleted = 0
148
+ SwallowedException.where("period_hour < ?", cutoff).in_batches(of: 1000) do |batch|
149
+ deleted += batch.delete_all
150
+ end
151
+
152
+ if deleted > 0
153
+ RailsErrorDashboard::Logger.info(
154
+ "[RailsErrorDashboard] Retention cleanup: deleted #{deleted} swallowed exception records"
155
+ )
156
+ end
157
+ rescue => e
158
+ RailsErrorDashboard::Logger.debug(
159
+ "[RailsErrorDashboard] Swallowed exception retention cleanup failed: #{e.class} - #{e.message}"
160
+ )
161
+ end
162
+
163
+ # Timing gaps are pruned by their OWN age, not by a group.
164
+ #
165
+ # A gap describes an interval, not an error, so there is no error_log_id to
166
+ # cascade from -- without this it would accumulate for the life of the
167
+ # installation. It also runs ABOVE the early return in #perform, which
168
+ # fires whenever no error logs happen to be expired: gaps expire
169
+ # independently of errors, exactly like the rack-attack events whose
170
+ # comment already warns about this.
171
+ #
172
+ # Safe to prune on covered_until: once the cutoff has moved past a gap, no
173
+ # window the dashboard displays can still reach it, so the warning it
174
+ # carries is no longer meaningful. Own rescue, like its siblings.
175
+ def cleanup_event_timing_gaps(cutoff)
176
+ return unless EventTimingGap.table_exists?
177
+
178
+ # Kept for the LONGER of the two horizons that govern it.
179
+ #
180
+ # This used the configured retention cutoff alone, which is wrong
181
+ # whenever retention is shorter than the window the dashboard reports on:
182
+ # at retention_days = 7 the gap was deleted while the events it qualified
183
+ # were still on the page -- ten monthly events, no warning, group still
184
+ # active. A gap outlives its own retention precisely because the figures
185
+ # it qualifies do.
186
+ #
187
+ # Deleting it only once BOTH clocks have passed means the warning can
188
+ # never disappear while the numbers it describes are still displayed.
189
+ # The reverse case is unaffected: with the 90-day default the retention
190
+ # cutoff is already the later of the two, so nothing is kept longer than
191
+ # before.
192
+ gap_cutoff = [ cutoff, Queries::DashboardStats::WIDEST_DISPLAYED_WINDOW.ago ].min
193
+
194
+ deleted = 0
195
+ EventTimingGap.where("covered_until < ?", gap_cutoff).in_batches(of: 1000) do |batch|
196
+ deleted += batch.delete_all
197
+ end
198
+
199
+ if deleted > 0
200
+ RailsErrorDashboard::Logger.info(
201
+ "[RailsErrorDashboard] Retention cleanup: deleted #{deleted} event timing gaps"
202
+ )
203
+ end
204
+ rescue => e
205
+ RailsErrorDashboard::Logger.debug(
206
+ "[RailsErrorDashboard] Event timing gap retention cleanup failed: #{e.class} - #{e.message}"
207
+ )
208
+ end
209
+
89
210
  # Expire aggregated Rack Attack event rows. Isolated in its own rescue so a
90
211
  # failure here (e.g. table not yet migrated) never blocks error cleanup.
91
212
  def cleanup_rack_attack_events(cutoff)
@@ -21,11 +21,14 @@ module RailsErrorDashboard
21
21
  entries: entries, overflow: overflow, episode: episode, batch_id: batch_id
22
22
  )
23
23
 
24
- # A batch that reconciled nothing because every write failed is not a
25
- # delivered batch. Fail the job so Active Job retries it rather than
26
- # dropping counts that were only ever held in one process's memory.
24
+ # A batch that wrote nothing is not a delivered batch. Fail the job so
25
+ # Active Job retries it rather than dropping counts that were only ever
26
+ # held in one process's memory. Two shapes reach here: every entry failed
27
+ # permanently, and a transient store failure that rolled the whole batch
28
+ # back (retryable: true) -- the latter is intact and safe to replay.
27
29
  if result.is_a?(Hash) && result[:success] == false
28
- raise FlushFailed, "storm flush reconciled nothing: #{result[:error]}"
30
+ reason = result[:retryable] ? "storm flush rolled back" : "storm flush reconciled nothing"
31
+ raise FlushFailed, "#{reason}: #{result[:error]}"
29
32
  end
30
33
 
31
34
  result
@@ -7,6 +7,8 @@ module RailsErrorDashboard
7
7
  # help nobody) — this one message replaces them. The gate guarantees at
8
8
  # most one enqueue per episode; this job just delivers.
9
9
  class StormNotificationJob < ApplicationJob
10
+ include Concerns::PlainChannelMessage
11
+
10
12
  queue_as :default
11
13
 
12
14
  # @param started_at [String] ISO8601 episode start
@@ -17,23 +19,12 @@ module RailsErrorDashboard
17
19
  config = RailsErrorDashboard.configuration
18
20
  message = build_message(started_at, state, config, job_locale(locale))
19
21
 
20
- if config.enable_slack_notifications && config.slack_webhook_url.present?
21
- post_json(config.slack_webhook_url, { text: message })
22
- end
23
-
24
- if config.enable_discord_notifications && config.discord_webhook_url.present?
25
- post_json(config.discord_webhook_url, { content: message })
26
- end
27
-
28
- if config.enable_webhook_notifications && config.webhook_urls.any?
29
- payload = {
30
- event: "error_storm_detected",
31
- started_at: started_at,
32
- state: state,
33
- application: app_name(config)
34
- }
35
- config.webhook_urls.each { |url| post_json(url, payload) }
36
- end
22
+ deliver_plain_message(message, {
23
+ event: "error_storm_detected",
24
+ started_at: started_at,
25
+ state: state,
26
+ application: app_name(config)
27
+ })
37
28
  rescue => e
38
29
  Rails.logger.error("[RailsErrorDashboard] StormNotificationJob failed: #{e.class} - #{e.message}")
39
30
  end
@@ -61,32 +52,5 @@ module RailsErrorDashboard
61
52
  "#{t("red.notifications.storm.engaged", locale, mode: mode)} " \
62
53
  "#{t("red.notifications.storm.suppressed", locale)}#{link}"
63
54
  end
64
-
65
- def t(key, locale, **options)
66
- RailsErrorDashboard::I18nStore.translate(key, locale: locale, **options)
67
- end
68
-
69
- def app_name(config)
70
- config.application_name || ENV["APPLICATION_NAME"] ||
71
- (defined?(Rails) && Rails.application.class.module_parent_name) || "Rails Application"
72
- end
73
-
74
- def post_json(url, payload)
75
- if defined?(HTTParty)
76
- HTTParty.post(url, body: payload.to_json,
77
- headers: { "Content-Type" => "application/json" }, timeout: 10)
78
- else
79
- uri = URI(url)
80
- http = Net::HTTP.new(uri.host, uri.port)
81
- http.use_ssl = uri.scheme == "https"
82
- http.open_timeout = 5
83
- http.read_timeout = 10
84
- request = Net::HTTP::Post.new(uri.path, { "Content-Type" => "application/json" })
85
- request.body = payload.to_json
86
- http.request(request)
87
- end
88
- rescue => e
89
- Rails.logger.error("[RailsErrorDashboard] Storm notification post failed: #{e.message}")
90
- end
91
55
  end
92
56
  end
@@ -48,17 +48,20 @@ module RailsErrorDashboard
48
48
  return nil if mean.nil? || std_dev.nil?
49
49
  return nil if current_count <= mean
50
50
 
51
- std_devs_above = (current_count - mean) / std_dev
51
+ # nil when std_dev is zero: a perfectly flat history has no spread to
52
+ # measure against, and dividing by it made every count :critical.
53
+ sigma = std_devs_above_mean(current_count)
54
+ return nil if sigma.nil?
52
55
 
53
- case std_devs_above
54
- when sensitivity..(sensitivity + 1)
55
- :elevated
56
- when (sensitivity + 1)..(sensitivity + 2)
57
- :high
58
- when (sensitivity + 2)..Float::INFINITY
56
+ # Half-open bands, highest first: a boundary value belongs to the band it
57
+ # opens (3.0 is :high, 4.0 is :critical). Inclusive ranges matched
58
+ # top-down gave each boundary to the band below it.
59
+ if sigma >= sensitivity + 2
59
60
  :critical
60
- else
61
- nil
61
+ elsif sigma >= sensitivity + 1
62
+ :high
63
+ elsif sigma >= sensitivity
64
+ :elevated
62
65
  end
63
66
  end
64
67
 
@@ -14,11 +14,6 @@ module RailsErrorDashboard
14
14
  scope :recent_first, -> { order(created_at: :desc) }
15
15
  scope :oldest_first, -> { order(created_at: :asc) }
16
16
 
17
- # Get formatted timestamp for display
18
- def formatted_time
19
- created_at.strftime("%b %d, %Y at %I:%M %p")
20
- end
21
-
22
17
  # Check if comment was created recently (within last hour)
23
18
  def recent?
24
19
  created_at > 1.hour.ago
@@ -37,6 +37,16 @@ module RailsErrorDashboard
37
37
  # Association for tracking individual error occurrences
38
38
  has_many :error_occurrences, class_name: "RailsErrorDashboard::ErrorOccurrence", dependent: :destroy
39
39
 
40
+ # Hour buckets for storm-shed events. delete_all, not destroy: these rows
41
+ # are pure counters with no callbacks, and a group can own one per hour.
42
+ #
43
+ # The association has to live HERE. `dependent:` on EventCount's own
44
+ # belongs_to is rejected by Rails (":dependent option must be one of
45
+ # [:destroy, :delete, :destroy_async]"), so the cleanup can only be
46
+ # declared from the parent side. Retention deletes them separately as well,
47
+ # because it uses delete_all, which does not fire callbacks.
48
+ has_many :event_counts, class_name: "RailsErrorDashboard::EventCount", dependent: :delete_all
49
+
40
50
  # Comments used as internal audit trail for workflow actions (snooze, mute, status changes).
41
51
  # Manual comment form removed in v0.6 — discussion now lives on issue tracker.
42
52
  has_many :comments, class_name: "RailsErrorDashboard::ErrorComment", foreign_key: :error_log_id, dependent: :destroy
@@ -86,9 +96,9 @@ module RailsErrorDashboard
86
96
  after_create_commit -> { Services::ErrorBroadcaster.broadcast_new(self) }
87
97
  after_update_commit -> { Services::ErrorBroadcaster.broadcast_update(self) }
88
98
 
89
- # Cache invalidation - clear analytics caches when errors are created/updated/deleted
90
- after_save -> { Services::AnalyticsCacheManager.clear }
91
- after_destroy -> { Services::AnalyticsCacheManager.clear }
99
+ # No cache invalidation here, on purpose. A save is what a CAPTURE does, in
100
+ # the host app's request thread; the stats caches expire by TTL instead, and
101
+ # the commands behind user actions call AnalyticsCacheManager.clear themselves.
92
102
 
93
103
  def set_defaults
94
104
  self.platform ||= "API"
@@ -271,7 +281,13 @@ module RailsErrorDashboard
271
281
  end
272
282
  end
273
283
 
284
+ # The five workflow statuses. One list, so that a command can tell
285
+ # "unknown status" from "known, but not reachable from here".
286
+ STATUSES = %w[new in_progress investigating resolved wont_fix].freeze
287
+
274
288
  def can_transition_to?(new_status)
289
+ return false unless STATUSES.include?(new_status)
290
+
275
291
  # Define valid status transitions
276
292
  valid_transitions = {
277
293
  "new" => [ "in_progress", "investigating", "wont_fix" ],
@@ -35,6 +35,23 @@ module RailsErrorDashboard
35
35
  # after the user's configuration is loaded
36
36
  # See lib/rails_error_dashboard/engine.rb
37
37
 
38
+ # Read-side half of the invalid-bytes defence. Capture scrubs what it
39
+ # writes, but SQLite and MySQL will happily hold a row written before that
40
+ # (or by anything else), and one such string takes down every page that
41
+ # renders it: ERB raises "invalid byte sequence in UTF-8". Scrubbing on load
42
+ # makes the page render whether or not the operator has run
43
+ # error_dashboard:scrub_invalid_encoding yet. In memory only: nothing is
44
+ # written, and the record is not left dirty.
45
+ after_find :scrub_invalid_strings
46
+
47
+ # Names of the attributes the load-time scrub had to repair, so that
48
+ # Commands::ScrubInvalidEncoding can persist the repair: once the values
49
+ # are clean in memory, re-reading them can no longer tell a bad row apart.
50
+ # @return [Array<String>]
51
+ def invalid_encoding_attributes
52
+ @invalid_encoding_attributes || []
53
+ end
54
+
38
55
  # Rails' default for a string column when the adapter has one (MySQL).
39
56
  DEFAULT_STRING_LIMIT = 255
40
57
 
@@ -68,5 +85,22 @@ module RailsErrorDashboard
68
85
  limits[name] = column.limit || DEFAULT_STRING_LIMIT if column.type == :string
69
86
  end
70
87
  end
88
+
89
+ private
90
+
91
+ def scrub_invalid_strings
92
+ @attributes.each_value do |attribute|
93
+ value = attribute.value_before_type_cast
94
+ next unless value.is_a?(String)
95
+ next if value.encoding == Encoding::UTF_8 && value.valid_encoding? && !value.include?("\0")
96
+
97
+ name = attribute.name
98
+ self[name] = Services::EncodingSanitizer.scrub(value)
99
+ clear_attribute_changes([ name ])
100
+ (@invalid_encoding_attributes ||= []) << name
101
+ end
102
+ rescue => e
103
+ RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] scrub on load failed: #{e.class}: #{e.message}")
104
+ end
71
105
  end
72
106
  end
@@ -23,7 +23,15 @@ module RailsErrorDashboard
23
23
  scope :in_time_window, ->(start_time, end_time) { where(occurred_at: start_time..end_time) }
24
24
  scope :for_user, ->(user_id) { where(user_id: user_id) }
25
25
  scope :for_request, ->(request_id) { where(request_id: request_id) }
26
- scope :for_session, ->(session_id) { where(session_id: session_id) }
26
+ # Takes the raw session ID (or a stored digest). Rows are stored as a keyed
27
+ # digest while filter_sensitive_data is on, and raw otherwise or before the
28
+ # upgrade, so both forms are matched. Blank matches nothing.
29
+ scope :for_session, lambda { |session_id|
30
+ raw = session_id.to_s
31
+ next none if raw.empty?
32
+
33
+ where(session_id: [ raw, Services::SensitiveDataFilter.digest_session_id(raw) ].compact.uniq)
34
+ }
27
35
 
28
36
  # Find occurrences within a time window around this occurrence
29
37
  # @param window_minutes [Integer] Time window in minutes (default: 5)
@@ -0,0 +1,132 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RailsErrorDashboard
4
+ # How many storm-shed events landed on one error group in one hour.
5
+ #
6
+ # Storm protection sheds events by folding N of them into the group's
7
+ # occurrence_count and writing no ErrorOccurrence row. That keeps the total
8
+ # exact but leaves the events with no timestamp of their own, so a window
9
+ # query has nothing to filter them by. This table gives them one.
10
+ #
11
+ # Window volume is therefore:
12
+ #
13
+ # ErrorOccurrence rows in the window + EventCount buckets in the window
14
+ #
15
+ # Ordinary captures contribute the first term, shed events the second, and
16
+ # neither is double counted: an event that wrote an occurrence row is never
17
+ # also bucketed here.
18
+ #
19
+ # Inherits ErrorLogsRecord so separate-database routing applies.
20
+ class EventCount < ErrorLogsRecord
21
+ self.table_name = "rails_error_dashboard_event_counts"
22
+
23
+ belongs_to :error_log, class_name: "RailsErrorDashboard::ErrorLog", optional: true
24
+
25
+ scope :in_window, ->(from, to = nil) {
26
+ scope = where(arel_table[:bucket_at].gteq(from))
27
+ to ? scope.where(arel_table[:bucket_at].lt(to)) : scope
28
+ }
29
+
30
+ # Bucket width, shared with the PRODUCER.
31
+ #
32
+ # This must equal CountBuffer::BUCKET_SECONDS, and there is a spec that
33
+ # asserts it. The producer tallies 15-minute buckets precisely so that a
34
+ # local midnight falls on a bucket EDGE in every UTC offset in use --
35
+ # including +05:30 (Kolkata) and +05:45 (Kathmandu). Rounding those buckets
36
+ # to the hour here destroyed exactly the property the producer paid for:
37
+ # a storm event at 23:59:30 and one at 00:00:30 in Kolkata shared one
38
+ # bucket, and a day's total was wrong by the whole straddle.
39
+ BUCKET_SECONDS = Services::StormProtection::CountBuffer::BUCKET_SECONDS
40
+
41
+ # The bucket a time belongs to, in UTC. One definition, used by the writer
42
+ # and the readers -- a mismatch here silently splits or merges a bucket.
43
+ # @param time [Time]
44
+ # @return [Time]
45
+ def self.bucket_for(time)
46
+ time = (time || Time.current)
47
+ Time.at((time.to_i / BUCKET_SECONDS) * BUCKET_SECONDS).utc
48
+ end
49
+
50
+ # Add +count+ shed events to (error_log_id, bucket_at), creating the row if
51
+ # it is not there yet.
52
+ #
53
+ # Adapter-portable by hand rather than via upsert_all: the increment has to
54
+ # read the existing value, and the ON CONFLICT / ON DUPLICATE KEY syntaxes
55
+ # differ. The UPDATE-first shape means the common case (a storm flushing
56
+ # repeatedly into the same hour) is a single statement, and the INSERT race
57
+ # is resolved by retrying the UPDATE once.
58
+ #
59
+ # @return [Symbol] :written when the bucket landed, :unavailable when the
60
+ # rollup is permanently unusable (no table, bad args, an adapter that
61
+ # refuses the statement). A TRANSIENT failure raises instead.
62
+ #
63
+ # Three states, not a boolean: the caller has to tell "the bucket is
64
+ # permanently unavailable, degrade" from "the write failed, retry the
65
+ # batch". Collapsing both into false made FlushStormCounts abort the
66
+ # whole transaction on a host that had simply never migrated the rollup
67
+ # table -- rolling back the authoritative lifetime count with it.
68
+ def self.accumulate(error_log_id:, bucket_at:, count:)
69
+ return :unavailable unless error_log_id && count.to_i.positive?
70
+ return :unavailable unless table_exists?
71
+
72
+ bucket = bucket_for(bucket_at)
73
+ updated = where(error_log_id: error_log_id, bucket_at: bucket)
74
+ .update_all([ "count = count + ?, updated_at = ?", count.to_i, Time.current ])
75
+ return :written if updated.positive?
76
+
77
+ begin
78
+ # requires_new: a failed INSERT aborts its transaction on PostgreSQL,
79
+ # which would take the caller's surrounding transaction with it.
80
+ transaction(requires_new: true) do
81
+ create!(error_log_id: error_log_id, bucket_at: bucket, count: count.to_i)
82
+ end
83
+ :written
84
+ rescue ActiveRecord::RecordNotUnique
85
+ # Another process created the same bucket between the UPDATE and the
86
+ # INSERT. The row exists now, so the UPDATE that missed a moment ago
87
+ # succeeds.
88
+ where(error_log_id: error_log_id, bucket_at: bucket)
89
+ .update_all([ "count = count + ?, updated_at = ?", count.to_i, Time.current ])
90
+ .positive? ? :written : :unavailable
91
+ end
92
+ rescue *Commands::LogError::RETRYABLE_STORE_ERRORS => e
93
+ # TRANSIENT: the store may be back in a moment. Do NOT swallow it.
94
+ #
95
+ # Returning false here let FlushStormCounts finalize its batch ledger
96
+ # with the bucket missing, so the replay was suppressed as
97
+ # already-applied and those events were erased from every time window --
98
+ # permanently -- while the lifetime count stayed correct. The caller
99
+ # turns this into an abort-and-retry of the whole batch.
100
+ RailsErrorDashboard::Logger.debug(
101
+ "[RailsErrorDashboard] EventCount.accumulate hit a transient failure: #{e.class} - #{e.message}"
102
+ )
103
+ raise
104
+ rescue StandardError => e
105
+ # PERMANENT: a malformed row, a missing table, an adapter that refuses
106
+ # this statement. Retrying cannot help, and failing the flush would turn
107
+ # a lost time bucket into a lost COUNT -- the authoritative total is
108
+ # occurrence_count on the group, and it is already written.
109
+ #
110
+ # This is the only case the old comment actually described.
111
+ RailsErrorDashboard::Logger.debug(
112
+ "[RailsErrorDashboard] EventCount.accumulate failed permanently: #{e.class} - #{e.message}"
113
+ )
114
+ :unavailable
115
+ end
116
+
117
+ # Whether the rollup table is usable.
118
+ #
119
+ # A transient connection failure here is NOT "the table does not exist":
120
+ # swallowing it returned false before any write was attempted, so a storm
121
+ # flush reported success with no bucket written, and EventVolume silently
122
+ # dropped the bucket term from its reads. Let the transient class through
123
+ # so the caller can retry; only a genuinely absent table returns false.
124
+ def self.table_exists?
125
+ connection.table_exists?(table_name)
126
+ rescue *Commands::LogError::RETRYABLE_STORE_ERRORS
127
+ raise
128
+ rescue StandardError
129
+ false
130
+ end
131
+ end
132
+ end
@@ -0,0 +1,55 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RailsErrorDashboard
4
+ # An interval whose per-event timing evidence was lost.
5
+ #
6
+ # See the migration for why this is its own table rather than a flag on the
7
+ # storm episode: the episode is optional, and it is written after the counts
8
+ # transaction has already committed.
9
+ #
10
+ # Inherits ErrorLogsRecord so separate-database routing applies.
11
+ class EventTimingGap < ErrorLogsRecord
12
+ self.table_name = "rails_error_dashboard_event_timing_gaps"
13
+
14
+ # Gaps overlapping [from, to). Open-ended when +to+ is nil.
15
+ #
16
+ # Overlap, not containment: a gap that began before the window and runs
17
+ # into it still makes that window's timing unreliable. Checking only
18
+ # "starts inside the window" is the mistake the episode predicate made.
19
+ scope :overlapping, ->(from, to = nil) {
20
+ scope = where(arel_table[:covered_until].gteq(from))
21
+ to ? scope.where(arel_table[:covered_from].lt(to)) : scope
22
+ }
23
+
24
+ # Whether any recorded gap affects the given window.
25
+ #
26
+ # Every guard is deliberate: the table may not be migrated on an older
27
+ # host, and this is read on the dashboard path, which must never raise.
28
+ #
29
+ # @param from [Time] start of the window being displayed
30
+ # @param application_id [Integer, nil]
31
+ # @return [Boolean]
32
+ def self.affecting?(from, application_id: nil)
33
+ return false unless table_exists?
34
+
35
+ scope = overlapping(from)
36
+ # A NULL application_id means the gap applies everywhere, so it is
37
+ # included whatever is being filtered for.
38
+ scope = scope.where(application_id: [ application_id, nil ]) if application_id.present?
39
+ scope.exists?
40
+ rescue StandardError => e
41
+ RailsErrorDashboard::Logger.debug(
42
+ "[RailsErrorDashboard] EventTimingGap.affecting? failed: #{e.class} - #{e.message}"
43
+ )
44
+ false
45
+ end
46
+
47
+ def self.table_exists?
48
+ connection.table_exists?(table_name)
49
+ rescue *Commands::LogError::RETRYABLE_STORE_ERRORS
50
+ raise
51
+ rescue StandardError
52
+ false
53
+ end
54
+ end
55
+ end