rails_error_dashboard 0.12.1 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. checksums.yaml +4 -4
  2. data/app/controllers/rails_error_dashboard/application_controller.rb +92 -11
  3. data/app/controllers/rails_error_dashboard/errors_controller.rb +111 -34
  4. data/app/controllers/rails_error_dashboard/webhooks_controller.rb +3 -0
  5. data/app/helpers/rails_error_dashboard/application_helper.rb +46 -1
  6. data/app/helpers/rails_error_dashboard/backtrace_helper.rb +8 -0
  7. data/app/jobs/rails_error_dashboard/async_error_logging_job.rb +10 -0
  8. data/app/jobs/rails_error_dashboard/concerns/plain_channel_message.rb +71 -0
  9. data/app/jobs/rails_error_dashboard/notification_burst_summary_job.rb +55 -0
  10. data/app/jobs/rails_error_dashboard/retention_cleanup_job.rb +122 -1
  11. data/app/jobs/rails_error_dashboard/storm_flush_job.rb +7 -4
  12. data/app/jobs/rails_error_dashboard/storm_notification_job.rb +8 -44
  13. data/app/models/rails_error_dashboard/error_baseline.rb +12 -9
  14. data/app/models/rails_error_dashboard/error_comment.rb +0 -5
  15. data/app/models/rails_error_dashboard/error_log.rb +19 -3
  16. data/app/models/rails_error_dashboard/error_logs_record.rb +34 -0
  17. data/app/models/rails_error_dashboard/error_occurrence.rb +9 -1
  18. data/app/models/rails_error_dashboard/event_count.rb +132 -0
  19. data/app/models/rails_error_dashboard/event_timing_gap.rb +55 -0
  20. data/app/views/layouts/rails_error_dashboard.html.erb +58 -5
  21. data/app/views/rails_error_dashboard/errors/_discussion.html.erb +18 -11
  22. data/app/views/rails_error_dashboard/errors/_issue_section.html.erb +22 -8
  23. data/app/views/rails_error_dashboard/errors/_request_context.html.erb +2 -0
  24. data/app/views/rails_error_dashboard/errors/_sidebar_metadata.html.erb +15 -6
  25. data/app/views/rails_error_dashboard/errors/analytics.html.erb +8 -8
  26. data/app/views/rails_error_dashboard/errors/diagnostic_dumps.html.erb +1 -1
  27. data/app/views/rails_error_dashboard/errors/index.html.erb +6 -1
  28. data/app/views/rails_error_dashboard/errors/overview.html.erb +12 -0
  29. data/app/views/rails_error_dashboard/errors/platform_comparison.html.erb +7 -7
  30. data/app/views/rails_error_dashboard/errors/releases.html.erb +2 -2
  31. data/app/views/rails_error_dashboard/errors/settings.html.erb +2 -0
  32. data/app/views/rails_error_dashboard/errors/show.html.erb +3 -3
  33. data/config/locales/de.yml +30 -0
  34. data/config/locales/en.yml +37 -0
  35. data/config/locales/es.yml +30 -0
  36. data/config/locales/fr.yml +30 -0
  37. data/config/locales/it.yml +30 -0
  38. data/config/locales/ja.yml +30 -0
  39. data/config/locales/pl.yml +30 -0
  40. data/config/locales/pt-BR.yml +30 -0
  41. data/config/locales/ru.yml +30 -0
  42. data/config/locales/uk.yml +30 -0
  43. data/config/locales/zh-CN.yml +30 -0
  44. data/db/migrate/20260917000001_add_last_notified_at_to_error_logs.rb +40 -0
  45. data/db/migrate/20260919000001_create_event_counts.rb +71 -0
  46. data/db/migrate/20260920000001_add_buckets_incomplete_to_storm_events.rb +25 -0
  47. data/db/migrate/20260920000002_create_event_timing_gaps.rb +55 -0
  48. data/lib/generators/rails_error_dashboard/install/templates/initializer.rb +12 -1
  49. data/lib/rails_error_dashboard/commands/assign_error.rb +12 -2
  50. data/lib/rails_error_dashboard/commands/backfill_environments.rb +2 -0
  51. data/lib/rails_error_dashboard/commands/backfill_resolved_at.rb +42 -0
  52. data/lib/rails_error_dashboard/commands/batch_delete_errors.rb +1 -0
  53. data/lib/rails_error_dashboard/commands/batch_mute_errors.rb +2 -0
  54. data/lib/rails_error_dashboard/commands/batch_resolve_errors.rb +2 -0
  55. data/lib/rails_error_dashboard/commands/batch_unmute_errors.rb +2 -0
  56. data/lib/rails_error_dashboard/commands/find_or_increment_error.rb +109 -11
  57. data/lib/rails_error_dashboard/commands/flush_rack_attack_events.rb +3 -1
  58. data/lib/rails_error_dashboard/commands/flush_storm_counts.rb +252 -12
  59. data/lib/rails_error_dashboard/commands/flush_swallowed_exceptions.rb +5 -2
  60. data/lib/rails_error_dashboard/commands/link_existing_issue.rb +1 -0
  61. data/lib/rails_error_dashboard/commands/log_error.rb +291 -40
  62. data/lib/rails_error_dashboard/commands/mute_error.rb +1 -0
  63. data/lib/rails_error_dashboard/commands/resolve_error.rb +2 -0
  64. data/lib/rails_error_dashboard/commands/scrub_invalid_encoding.rb +104 -0
  65. data/lib/rails_error_dashboard/commands/snooze_error.rb +34 -10
  66. data/lib/rails_error_dashboard/commands/unmute_error.rb +1 -0
  67. data/lib/rails_error_dashboard/commands/update_error_priority.rb +23 -2
  68. data/lib/rails_error_dashboard/commands/update_error_status.rb +33 -6
  69. data/lib/rails_error_dashboard/configuration.rb +39 -1
  70. data/lib/rails_error_dashboard/engine.rb +28 -0
  71. data/lib/rails_error_dashboard/manual_error_reporter.rb +16 -5
  72. data/lib/rails_error_dashboard/queries/analytics_stats.rb +89 -30
  73. data/lib/rails_error_dashboard/queries/baseline_stats.rb +107 -0
  74. data/lib/rails_error_dashboard/queries/dashboard_stats.rb +216 -74
  75. data/lib/rails_error_dashboard/queries/error_correlation.rb +6 -3
  76. data/lib/rails_error_dashboard/queries/errors_list.rb +15 -2
  77. data/lib/rails_error_dashboard/queries/event_volume.rb +503 -0
  78. data/lib/rails_error_dashboard/queries/similar_errors.rb +1 -1
  79. data/lib/rails_error_dashboard/services/analytics_cache_manager.rb +43 -18
  80. data/lib/rails_error_dashboard/services/backtrace_processor.rb +3 -1
  81. data/lib/rails_error_dashboard/services/breadcrumb_collector.rb +23 -0
  82. data/lib/rails_error_dashboard/services/cause_chain_extractor.rb +3 -1
  83. data/lib/rails_error_dashboard/services/codeberg_issue_client.rb +13 -2
  84. data/lib/rails_error_dashboard/services/diagnostic_dump_generator.rb +5 -3
  85. data/lib/rails_error_dashboard/services/encoding_sanitizer.rb +80 -0
  86. data/lib/rails_error_dashboard/services/error_broadcaster.rb +180 -32
  87. data/lib/rails_error_dashboard/services/error_hash_generator.rb +9 -3
  88. data/lib/rails_error_dashboard/services/error_notification_dispatcher.rb +13 -0
  89. data/lib/rails_error_dashboard/services/exception_filter.rb +55 -0
  90. data/lib/rails_error_dashboard/services/git_head_reader.rb +102 -0
  91. data/lib/rails_error_dashboard/services/notification_throttler.rb +172 -28
  92. data/lib/rails_error_dashboard/services/sensitive_data_filter.rb +47 -1
  93. data/lib/rails_error_dashboard/services/storm_protection/circuit_breaker.rb +59 -3
  94. data/lib/rails_error_dashboard/services/storm_protection/count_buffer.rb +64 -6
  95. data/lib/rails_error_dashboard/services/storm_protection/fingerprint_buckets.rb +25 -2
  96. data/lib/rails_error_dashboard/services/storm_protection/gate.rb +32 -5
  97. data/lib/rails_error_dashboard/services/swallowed_exception_tracker.rb +99 -23
  98. data/lib/rails_error_dashboard/services/url_safety.rb +40 -0
  99. data/lib/rails_error_dashboard/services/variable_serializer.rb +125 -10
  100. data/lib/rails_error_dashboard/subscribers/breadcrumb_subscriber.rb +111 -0
  101. data/lib/rails_error_dashboard/subscribers/issue_tracker_subscriber.rb +11 -2
  102. data/lib/rails_error_dashboard/value_objects/error_context.rb +40 -3
  103. data/lib/rails_error_dashboard/version.rb +1 -1
  104. data/lib/rails_error_dashboard.rb +34 -0
  105. data/lib/tasks/error_dashboard.rake +54 -4
  106. metadata +16 -2
@@ -5,6 +5,13 @@ module RailsErrorDashboard
5
5
  # Query: Fetch dashboard statistics
6
6
  # This is a read operation that aggregates error data for the dashboard
7
7
  class DashboardStats
8
+ # The widest window any figure on the Overview covers (total_month).
9
+ # The completeness predicate is matched to this, not to "today".
10
+ WIDEST_DISPLAYED_WINDOW = 30.days
11
+
12
+ # One instance answers ONE call: several aggregates (today's event count,
13
+ # the 7-day trend, spike detection) are memoised on it so that they are
14
+ # computed once per call rather than once per card that shows them.
8
15
  def initialize(application_id: nil)
9
16
  @application_id = application_id
10
17
  end
@@ -24,9 +31,9 @@ module RailsErrorDashboard
24
31
  # rows reported five users hitting one error as "1 error today".
25
32
  # Summing is also exact during a storm: counted-only events never
26
33
  # create occurrence rows, but they DO raise occurrence_count.
27
- total_today: event_count_since(Time.current.beginning_of_day),
28
- total_week: event_count_since(7.days.ago),
29
- total_month: event_count_since(30.days.ago),
34
+ total_today: today_event_count,
35
+ total_week: week_event_counts.values.sum,
36
+ total_month: month_event_count,
30
37
  unresolved: base_scope.unresolved.count,
31
38
  resolved: base_scope.resolved.count,
32
39
  reopened: reopened_count,
@@ -51,6 +58,13 @@ module RailsErrorDashboard
51
58
  # window, the dimension is incomplete and the page says so
52
59
  # rather than presenting an undercount as fact.
53
60
  affected_users_incomplete: affected_users_incomplete?,
61
+ # A storm episode in this window lost its per-bucket TIMING
62
+ # evidence, so every time-window figure for it rests on the
63
+ # group's own occurred_at rather than on per-event records.
64
+ # Reported for the same reason as affected_users_incomplete:
65
+ # an incomplete dimension should say so rather than present a
66
+ # figure of unknown completeness as fact.
67
+ event_timing_incomplete: event_timing_incomplete?,
54
68
  data_unavailable: false
55
69
  }
56
70
  end
@@ -68,6 +82,7 @@ module RailsErrorDashboard
68
82
  {
69
83
  data_unavailable: true,
70
84
  affected_users_incomplete: false,
85
+ event_timing_incomplete: false,
71
86
  total_today: 0,
72
87
  total_week: 0,
73
88
  total_month: 0,
@@ -93,13 +108,14 @@ module RailsErrorDashboard
93
108
  end
94
109
 
95
110
  def cache_key
96
- # Cache key includes last error update timestamp for auto-invalidation
97
- # Also includes current hour to ensure fresh data
98
- # Uses base_scope to respect application_id filter for proper cache isolation
111
+ # The cache GENERATION, not maximum(:updated_at): the timestamp cost a
112
+ # query per key build and changed on every capture, so the cache never
113
+ # hit while errors were arriving. Freshness after a capture is the
114
+ # 1-minute TTL; user actions bump the generation (AnalyticsCacheManager).
99
115
  [
100
116
  "dashboard_stats",
101
117
  @application_id || "all",
102
- base_scope.maximum(:updated_at)&.to_i || 0,
118
+ Services::AnalyticsCacheManager.generation,
103
119
  Time.current.hour
104
120
  ].join("/")
105
121
  end
@@ -112,14 +128,29 @@ module RailsErrorDashboard
112
128
  scope
113
129
  end
114
130
 
115
- # Total EVENTS in a window: the sum of every matching group's
116
- # occurrence_count. platform_comparison.rb counts the same way.
131
+ # Total EVENTS in a window -- the events that HAPPENED in it, not the
132
+ # lifetime volume of the groups first seen in it.
133
+ #
134
+ # This used to filter groups by occurred_at and sum occurrence_count.
135
+ # occurred_at is first-seen and is never rewritten on recurrence, so an
136
+ # error first seen at 23:59 that recurred at 00:01 reported zero errors
137
+ # today and two yesterday. Queries::EventVolume counts per-event records
138
+ # instead: occurrence rows, plus storm-shed time buckets, plus the
139
+ # remainder of any group that has neither.
117
140
  def event_count_since(since)
118
- base_scope.where("occurred_at >= ?", since).sum(:occurrence_count)
141
+ Queries::EventVolume.in_window(base_scope, since)
142
+ end
143
+
144
+ # The widest-window total, memoised: the stats hash shows it and the
145
+ # completeness predicate needs the same number. One instance answers one
146
+ # call, and this runs on the capture path via the stats broadcast, so a
147
+ # second identical aggregate here is pure waste.
148
+ def month_event_count
149
+ @month_event_count ||= event_count_since(WIDEST_DISPLAYED_WINDOW.ago)
119
150
  end
120
151
 
121
152
  def event_count_between(from, to)
122
- base_scope.where("occurred_at >= ? AND occurred_at < ?", from, to).sum(:occurrence_count)
153
+ Queries::EventVolume.in_window(base_scope, from, to)
123
154
  end
124
155
 
125
156
  # Occurrence rows carry the user of EACH event. The group's user_id is
@@ -147,67 +178,171 @@ module RailsErrorDashboard
147
178
  recorded = occurrence_scope
148
179
  .where("#{ErrorOccurrence.table_name}.occurred_at >= ?", Time.current.beginning_of_day)
149
180
  .count
150
- recorded < event_count_since(Time.current.beginning_of_day)
181
+ recorded < today_event_count
151
182
  rescue StandardError
152
183
  false
153
184
  end
154
185
 
186
+ # True when a storm episode overlapping this window degraded without
187
+ # writing its time buckets.
188
+ #
189
+ # Read from the persisted episode rather than recomputed: the flush that
190
+ # knew is long over by the time the dashboard renders, and the flag has
191
+ # to survive a replay. Every guard here is deliberate -- the storm table
192
+ # may not exist, and the column may not be migrated yet on an older host.
193
+ # True when any recorded timing gap overlaps a window this page shows.
194
+ #
195
+ # The page displays today, 7-day AND 30-day figures, so the predicate has
196
+ # to match the WIDEST of them. An earlier version asked only about today
197
+ # and dropped the warning while the weekly and monthly numbers on the
198
+ # same screen still contained the affected events.
199
+ #
200
+ # Read from EventTimingGap rather than the storm episode: the episode is
201
+ # optional (the gate can shed with its breaker closed), and it is written
202
+ # after the counts transaction commits. See the migration for the full
203
+ # history of why this moved.
204
+ def event_timing_incomplete?
205
+ # Nothing left to qualify, so no warning.
206
+ #
207
+ # Gaps are deliberately retained past their own retention while the
208
+ # figures they describe are still shown (see the cleanup job), so a gap
209
+ # can outlive every event it covered -- a group expiring is what
210
+ # removes those events. Warning about a window whose subject matter is
211
+ # gone is noise, and noise trains people to ignore the banner.
212
+ #
213
+ # Asked of surviving GROUPS, not of the event aggregate. The aggregate
214
+ # is the wrong witness here: losing the time buckets is exactly what
215
+ # makes events invisible to it, so an old group that recurs today
216
+ # reports zero for every window (EventVolume can only place an
217
+ # untracked remainder at the group's own occurred_at, which precedes
218
+ # the window -- see untracked_groups) while the events are real, the
219
+ # group is alive and a current gap records them. Reading that zero as
220
+ # "nothing happened" suppressed the banner in precisely the state it
221
+ # exists to announce.
222
+ #
223
+ # last_seen_at is the right evidence because it is what retention
224
+ # deletes on: a group is expired only once it has not been seen for
225
+ # retention_days, so "no group seen in this window" is the same fact as
226
+ # "the covered events are gone" -- and it is indexed, and immune to the
227
+ # timing loss itself.
228
+ #
229
+ # Suppressed only when BOTH witnesses are silent, because neither alone
230
+ # is sufficient and they fail in opposite directions. The aggregate
231
+ # misses an old group's untracked recurrence (the defect above); the
232
+ # liveness check misses the converse, since EventVolume windows
233
+ # occurrence rows and buckets on THEIR OWN timestamps against an
234
+ # unwindowed group set (see group_ids), so a group whose last_seen_at
235
+ # has fallen behind its own event rows can still put events on the page.
236
+ # Requiring both to be empty means the banner can never be dropped
237
+ # while any displayed figure is non-zero, which is the F22 invariant.
238
+ return false unless groups_seen_in_widest_window? || month_event_count.positive?
239
+
240
+ EventTimingGap.affecting?(WIDEST_DISPLAYED_WINDOW.ago, application_id: @application_id)
241
+ rescue StandardError
242
+ false
243
+ end
244
+
245
+ # Whether any error group was last seen inside the widest displayed
246
+ # window. Existence, not a count: one indexed row settles it, and this
247
+ # runs on the capture path via the stats broadcast.
248
+ #
249
+ # The NULL arm covers rows written before last_seen_at existed, whose
250
+ # occurred_at is the only timestamp they have -- the same COALESCE
251
+ # equivalence the retention job documents, written so each arm keeps its
252
+ # own index.
253
+ def groups_seen_in_widest_window?
254
+ cutoff = WIDEST_DISPLAYED_WINDOW.ago
255
+ base_scope.where(
256
+ "last_seen_at >= :cutoff OR (last_seen_at IS NULL AND occurred_at >= :cutoff)", cutoff: cutoff
257
+ ).exists?
258
+ end
259
+
155
260
  def reopened_count
156
261
  return 0 unless ErrorLog.column_names.include?("reopened_at")
157
262
 
158
263
  base_scope.where.not(reopened_at: nil).count
159
264
  end
160
265
 
266
+ # Top error types in the last 7 days, by EVENTS.
267
+ #
268
+ # Filtering groups by first-seen and summing their LIFETIME
269
+ # occurrence_count answered a different question: an August group that
270
+ # recurred today was excluded entirely, so reopening it produced an empty
271
+ # list while the headline total (already on EventVolume) counted the
272
+ # event. Routed through the same primitive as the total so the page's
273
+ # parts and its whole cannot drift apart.
161
274
  def top_errors
162
- base_scope.where("occurred_at >= ?", 7.days.ago)
163
- .group(:error_type)
164
- .sum(:occurrence_count)
165
- .sort_by { |_, count| -count }
166
- .first(10)
167
- .to_h
275
+ weekly_volume.by_group_attribute(:error_type)
276
+ .sort_by { |_, count| -count }
277
+ .first(10)
278
+ .to_h
279
+ end
280
+
281
+ # One EventVolume for the 7-day window, shared by every weekly breakdown.
282
+ def weekly_volume
283
+ @weekly_volume ||= Queries::EventVolume.new(base_scope, 7.days.ago)
168
284
  end
169
285
 
170
286
  # Get 7-day error trend (daily counts)
287
+ # Events per day, placed on the day they HAPPENED. Grouping the ErrorLog
288
+ # table by its own occurred_at put a group's whole lifetime volume on the
289
+ # day it was first seen, so a recurrence never moved the trend.
290
+ #
291
+ # The zero-filled date range is preserved: the chart needs a point for
292
+ # every day in the window, not only the days that had errors.
171
293
  def errors_trend_7d
172
- base_scope.where("occurred_at >= ?", 7.days.ago)
173
- .group_by_day(:occurred_at, range: 7.days.ago.to_date..Date.current, default_value: 0)
174
- .sum(:occurrence_count)
294
+ week_event_counts
175
295
  end
176
296
 
177
297
  # Get error counts by severity for last 7 days
178
298
  # OPTIMIZED: Use database filtering instead of loading all records into Ruby
299
+ # Weekly events by severity.
300
+ #
301
+ # Folded from the SAME per-type breakdown top_errors uses, rather than
302
+ # four independently scoped sums. Two reasons: it was summing LIFETIME
303
+ # counts of groups born in the window (so a reopened August group scored
304
+ # zero across every severity), and deriving both figures from one
305
+ # breakdown makes them agree by construction instead of by luck --
306
+ # asserted by spec/queries/dashboard_breakdowns_event_volume_spec.rb.
179
307
  def errors_by_severity_7d
180
- scoped_errors = base_scope.where("occurred_at >= ?", 7.days.ago)
181
-
182
- {
183
- critical: scoped_errors.where(error_type: Services::SeverityClassifier::CRITICAL_ERROR_TYPES).sum(:occurrence_count),
184
- high: scoped_errors.where(error_type: Services::SeverityClassifier::HIGH_SEVERITY_ERROR_TYPES).sum(:occurrence_count),
185
- medium: scoped_errors.where(error_type: Services::SeverityClassifier::MEDIUM_SEVERITY_ERROR_TYPES).sum(:occurrence_count),
186
- low: scoped_errors.where.not(
187
- error_type: Services::SeverityClassifier::CRITICAL_ERROR_TYPES +
188
- Services::SeverityClassifier::HIGH_SEVERITY_ERROR_TYPES +
189
- Services::SeverityClassifier::MEDIUM_SEVERITY_ERROR_TYPES
190
- ).sum(:occurrence_count)
191
- }
308
+ critical = Services::SeverityClassifier::CRITICAL_ERROR_TYPES
309
+ high = Services::SeverityClassifier::HIGH_SEVERITY_ERROR_TYPES
310
+ medium = Services::SeverityClassifier::MEDIUM_SEVERITY_ERROR_TYPES
311
+
312
+ totals = { critical: 0, high: 0, medium: 0, low: 0 }
313
+ weekly_volume.by_group_attribute(:error_type).each do |error_type, count|
314
+ bucket =
315
+ if critical.include?(error_type) then :critical
316
+ elsif high.include?(error_type) then :high
317
+ elsif medium.include?(error_type) then :medium
318
+ else :low
319
+ end
320
+ totals[bucket] += count
321
+ end
322
+ totals
192
323
  end
193
324
 
194
325
  # Detect if there's an error spike
195
326
  # Uses baselines if available, falls back to simple 2x average
327
+ #
328
+ # Memoised: the stats hash asks twice (spike_detected and spike_info), and
329
+ # this runs from the live stats broadcast inside the capture path.
196
330
  def spike_detected?
197
- return false if errors_trend_7d.empty?
331
+ return @spike_detected if defined?(@spike_detected)
198
332
 
199
- today_count = event_count_since(Time.current.beginning_of_day)
333
+ @spike_detected = compute_spike_detected
334
+ end
200
335
 
201
- # Try baseline-based detection first
202
- if baseline_anomaly_detected?(today_count)
203
- return true
204
- end
336
+ def compute_spike_detected
337
+ trend = errors_trend_7d
338
+ return false if trend.empty?
339
+ return true if baseline_anomalies.any?
205
340
 
206
341
  # Fall back to simple 2x average detection
207
- avg_count = errors_trend_7d.values.sum / 7.0
342
+ avg_count = trend.values.sum / 7.0
208
343
  return false if avg_count.zero?
209
344
 
210
- today_count >= (avg_count * 2)
345
+ today_event_count >= (avg_count * 2)
211
346
  end
212
347
 
213
348
  # Get spike information
@@ -215,7 +350,7 @@ module RailsErrorDashboard
215
350
  def spike_info
216
351
  return nil unless spike_detected?
217
352
 
218
- today_count = event_count_since(Time.current.beginning_of_day)
353
+ today_count = today_event_count
219
354
  avg_count = (errors_trend_7d.values.sum / 7.0).round(1)
220
355
 
221
356
  info = {
@@ -226,46 +361,46 @@ module RailsErrorDashboard
226
361
  }
227
362
 
228
363
  # Add baseline info if available
229
- baseline_info = baseline_anomaly_info(today_count)
364
+ baseline_info = baseline_anomaly_info
230
365
  info.merge!(baseline_info) if baseline_info.present?
231
366
 
232
367
  info
233
368
  end
234
369
 
235
- # Check if baseline indicates anomaly
236
- def baseline_anomaly_detected?(_count)
237
- return false unless defined?(Queries::BaselineStats)
238
-
239
- # Check most common error types for anomalies
240
- base_scope.distinct.pluck(:error_type, :platform).compact.any? do |(error_type, platform)|
241
- Queries::BaselineStats.new(error_type, platform)
242
- .check_current_anomaly(sensitivity: 2, application_id: @application_id)[:anomaly]
370
+ # Today, yesterday and the 7-day total all come from ONE daily breakdown.
371
+ # Each window is three queries (occurrence rows, shed buckets, untracked
372
+ # remainder), and this method runs on the capture path via the stats
373
+ # broadcast, so asking per window multiplied the query count.
374
+ def week_event_counts
375
+ @week_event_counts ||= begin
376
+ counts = Queries::EventVolume.by_day(base_scope, 7.days.ago)
377
+ (7.days.ago.to_date..Date.current).to_h { |day| [ day, counts[day] || 0 ] }
243
378
  end
244
379
  end
245
380
 
246
- # Get baseline anomaly information
247
- def baseline_anomaly_info(_total_count)
248
- return nil unless defined?(Queries::BaselineStats)
381
+ def today_event_count
382
+ @today_event_count ||= week_event_counts[Date.current].to_i
383
+ end
249
384
 
250
- # Find the most anomalous error type
251
- anomalies = base_scope.distinct.pluck(:error_type, :platform).compact.map do |(error_type, platform)|
252
- result = Queries::BaselineStats.new(error_type, platform)
253
- .check_current_anomaly(sensitivity: 2, application_id: @application_id)
254
- next unless result[:anomaly]
385
+ # Every anomalous (error_type, platform) pair, from a fixed number of
386
+ # queries (BaselineStats.current_anomalies), loaded once per call. This
387
+ # used to be about six queries per distinct pair, run twice.
388
+ def baseline_anomalies
389
+ return @baseline_anomalies if defined?(@baseline_anomalies)
255
390
 
256
- {
257
- error_type: error_type,
258
- platform: platform,
259
- count: result[:current_count],
260
- level: result[:level],
261
- std_devs_above: result[:std_devs_above]
262
- }
263
- end.compact
391
+ @baseline_anomalies = if defined?(Queries::BaselineStats)
392
+ Queries::BaselineStats.current_anomalies(sensitivity: 2, application_id: @application_id)
393
+ else
394
+ []
395
+ end
396
+ end
264
397
 
265
- return nil if anomalies.empty?
398
+ # Get baseline anomaly information
399
+ def baseline_anomaly_info
400
+ return nil if baseline_anomalies.empty?
266
401
 
267
402
  # Return info about worst anomaly
268
- worst = anomalies.max_by { |a| a[:std_devs_above] || 0 }
403
+ worst = baseline_anomalies.max_by { |a| a[:std_devs_above] || 0 }
269
404
  {
270
405
  baseline_detected: true,
271
406
  anomaly_error_type: worst[:error_type],
@@ -283,7 +418,7 @@ module RailsErrorDashboard
283
418
  # make a real failure percentage from, so the honest figure is the rate
284
419
  # itself, uncapped, labelled with its unit.
285
420
  def error_rate
286
- today_events = event_count_since(Time.current.beginning_of_day)
421
+ today_events = today_event_count
287
422
  return 0.0 if today_events.zero?
288
423
 
289
424
  hours_today = ((Time.current - Time.current.beginning_of_day) / 1.hour).round(1)
@@ -299,11 +434,12 @@ module RailsErrorDashboard
299
434
  # affected user. Storm count-only events create no occurrence row, so
300
435
  # this is a floor during a storm -- affected_users_incomplete? says when.
301
436
  def affected_users_today
302
- distinct_affected_users(Time.current.beginning_of_day, nil)
437
+ @affected_users_today ||= distinct_affected_users(Time.current.beginning_of_day, nil)
303
438
  end
304
439
 
305
440
  def affected_users_yesterday
306
- distinct_affected_users(1.day.ago.beginning_of_day, Time.current.beginning_of_day)
441
+ @affected_users_yesterday ||=
442
+ distinct_affected_users(1.day.ago.beginning_of_day, Time.current.beginning_of_day)
307
443
  end
308
444
 
309
445
  def distinct_affected_users(from, to)
@@ -334,8 +470,14 @@ module RailsErrorDashboard
334
470
 
335
471
  # Calculate percentage change in errors (today vs yesterday)
336
472
  def trend_percentage
337
- today = event_count_since(Time.current.beginning_of_day)
338
- yesterday = event_count_between(1.day.ago.beginning_of_day, Time.current.beginning_of_day)
473
+ return @trend_percentage if defined?(@trend_percentage)
474
+
475
+ @trend_percentage = compute_trend_percentage
476
+ end
477
+
478
+ def compute_trend_percentage
479
+ today = today_event_count
480
+ yesterday = week_event_counts[Date.current - 1].to_i
339
481
 
340
482
  return 0.0 if today.zero? && yesterday.zero?
341
483
  return 100.0 if yesterday.zero? && today.positive?
@@ -224,12 +224,15 @@ module RailsErrorDashboard
224
224
  previous_start = @start_date
225
225
  previous_end = current_start
226
226
 
227
- current_errors = ErrorLog
227
+ # Both periods come from base_query, which carries the application
228
+ # filter (and the window start). Querying ErrorLog directly made this
229
+ # the one panel on an application-filtered page that counted every app.
230
+ current_errors = base_query
228
231
  .where("occurred_at >= ?", current_start)
229
232
  .count
230
233
 
231
- previous_errors = ErrorLog
232
- .where("occurred_at >= ? AND occurred_at < ?", previous_start, previous_end)
234
+ previous_errors = base_query
235
+ .where("occurred_at < ?", previous_end)
233
236
  .count
234
237
 
235
238
  change_percentage = if previous_errors > 0
@@ -5,6 +5,9 @@ module RailsErrorDashboard
5
5
  # Query: Fetch errors with filtering and pagination
6
6
  # This is a read operation that returns a filtered collection of errors
7
7
  class ErrorsList
8
+ # Escape character for LIKE patterns; see filter_by_search.
9
+ LIKE_ESCAPE = "!"
10
+
8
11
  def self.call(filters = {})
9
12
  new(filters).call
10
13
  end
@@ -129,9 +132,19 @@ module RailsErrorDashboard
129
132
  else
130
133
  # Fall back to LIKE for SQLite/MySQL - search across all relevant fields
131
134
  # Use LOWER() for case-insensitive search
132
- search_pattern = "%#{@filters[:search]}%"
135
+ #
136
+ # % and _ in the term are escaped so they match themselves: a search
137
+ # for "snake_case" must not match "snakeXcase", and "%" must not
138
+ # match every row. The escape character is named explicitly because
139
+ # SQLite has no default one, and it is "!" rather than a backslash
140
+ # because a backslash inside a quoted SQL literal means different
141
+ # things to MySQL and to everyone else.
142
+ escaped = ActiveRecord::Base.sanitize_sql_like(@filters[:search].to_s, LIKE_ESCAPE)
143
+ search_pattern = "%#{escaped}%"
133
144
  query.where(
134
- "LOWER(message) LIKE LOWER(?) OR LOWER(COALESCE(backtrace, '')) LIKE LOWER(?) OR LOWER(error_type) LIKE LOWER(?)",
145
+ "LOWER(message) LIKE LOWER(?) ESCAPE '#{LIKE_ESCAPE}' " \
146
+ "OR LOWER(COALESCE(backtrace, '')) LIKE LOWER(?) ESCAPE '#{LIKE_ESCAPE}' " \
147
+ "OR LOWER(error_type) LIKE LOWER(?) ESCAPE '#{LIKE_ESCAPE}'",
135
148
  search_pattern, search_pattern, search_pattern
136
149
  )
137
150
  end