rails_error_dashboard 0.12.1 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. checksums.yaml +4 -4
  2. data/app/controllers/rails_error_dashboard/application_controller.rb +92 -11
  3. data/app/controllers/rails_error_dashboard/errors_controller.rb +111 -34
  4. data/app/controllers/rails_error_dashboard/webhooks_controller.rb +3 -0
  5. data/app/helpers/rails_error_dashboard/application_helper.rb +46 -1
  6. data/app/helpers/rails_error_dashboard/backtrace_helper.rb +8 -0
  7. data/app/jobs/rails_error_dashboard/async_error_logging_job.rb +10 -0
  8. data/app/jobs/rails_error_dashboard/concerns/plain_channel_message.rb +71 -0
  9. data/app/jobs/rails_error_dashboard/notification_burst_summary_job.rb +55 -0
  10. data/app/jobs/rails_error_dashboard/retention_cleanup_job.rb +122 -1
  11. data/app/jobs/rails_error_dashboard/storm_flush_job.rb +7 -4
  12. data/app/jobs/rails_error_dashboard/storm_notification_job.rb +8 -44
  13. data/app/models/rails_error_dashboard/error_baseline.rb +12 -9
  14. data/app/models/rails_error_dashboard/error_comment.rb +0 -5
  15. data/app/models/rails_error_dashboard/error_log.rb +19 -3
  16. data/app/models/rails_error_dashboard/error_logs_record.rb +34 -0
  17. data/app/models/rails_error_dashboard/error_occurrence.rb +9 -1
  18. data/app/models/rails_error_dashboard/event_count.rb +132 -0
  19. data/app/models/rails_error_dashboard/event_timing_gap.rb +55 -0
  20. data/app/views/layouts/rails_error_dashboard.html.erb +58 -5
  21. data/app/views/rails_error_dashboard/errors/_discussion.html.erb +18 -11
  22. data/app/views/rails_error_dashboard/errors/_issue_section.html.erb +22 -8
  23. data/app/views/rails_error_dashboard/errors/_request_context.html.erb +2 -0
  24. data/app/views/rails_error_dashboard/errors/_sidebar_metadata.html.erb +15 -6
  25. data/app/views/rails_error_dashboard/errors/analytics.html.erb +8 -8
  26. data/app/views/rails_error_dashboard/errors/diagnostic_dumps.html.erb +1 -1
  27. data/app/views/rails_error_dashboard/errors/index.html.erb +6 -1
  28. data/app/views/rails_error_dashboard/errors/overview.html.erb +12 -0
  29. data/app/views/rails_error_dashboard/errors/platform_comparison.html.erb +7 -7
  30. data/app/views/rails_error_dashboard/errors/releases.html.erb +2 -2
  31. data/app/views/rails_error_dashboard/errors/settings.html.erb +2 -0
  32. data/app/views/rails_error_dashboard/errors/show.html.erb +3 -3
  33. data/config/locales/de.yml +30 -0
  34. data/config/locales/en.yml +37 -0
  35. data/config/locales/es.yml +30 -0
  36. data/config/locales/fr.yml +30 -0
  37. data/config/locales/it.yml +30 -0
  38. data/config/locales/ja.yml +30 -0
  39. data/config/locales/pl.yml +30 -0
  40. data/config/locales/pt-BR.yml +30 -0
  41. data/config/locales/ru.yml +30 -0
  42. data/config/locales/uk.yml +30 -0
  43. data/config/locales/zh-CN.yml +30 -0
  44. data/db/migrate/20260917000001_add_last_notified_at_to_error_logs.rb +40 -0
  45. data/db/migrate/20260919000001_create_event_counts.rb +71 -0
  46. data/db/migrate/20260920000001_add_buckets_incomplete_to_storm_events.rb +25 -0
  47. data/db/migrate/20260920000002_create_event_timing_gaps.rb +55 -0
  48. data/lib/generators/rails_error_dashboard/install/templates/initializer.rb +12 -1
  49. data/lib/rails_error_dashboard/commands/assign_error.rb +12 -2
  50. data/lib/rails_error_dashboard/commands/backfill_environments.rb +2 -0
  51. data/lib/rails_error_dashboard/commands/backfill_resolved_at.rb +42 -0
  52. data/lib/rails_error_dashboard/commands/batch_delete_errors.rb +1 -0
  53. data/lib/rails_error_dashboard/commands/batch_mute_errors.rb +2 -0
  54. data/lib/rails_error_dashboard/commands/batch_resolve_errors.rb +2 -0
  55. data/lib/rails_error_dashboard/commands/batch_unmute_errors.rb +2 -0
  56. data/lib/rails_error_dashboard/commands/find_or_increment_error.rb +109 -11
  57. data/lib/rails_error_dashboard/commands/flush_rack_attack_events.rb +3 -1
  58. data/lib/rails_error_dashboard/commands/flush_storm_counts.rb +252 -12
  59. data/lib/rails_error_dashboard/commands/flush_swallowed_exceptions.rb +5 -2
  60. data/lib/rails_error_dashboard/commands/link_existing_issue.rb +1 -0
  61. data/lib/rails_error_dashboard/commands/log_error.rb +291 -40
  62. data/lib/rails_error_dashboard/commands/mute_error.rb +1 -0
  63. data/lib/rails_error_dashboard/commands/resolve_error.rb +2 -0
  64. data/lib/rails_error_dashboard/commands/scrub_invalid_encoding.rb +104 -0
  65. data/lib/rails_error_dashboard/commands/snooze_error.rb +34 -10
  66. data/lib/rails_error_dashboard/commands/unmute_error.rb +1 -0
  67. data/lib/rails_error_dashboard/commands/update_error_priority.rb +23 -2
  68. data/lib/rails_error_dashboard/commands/update_error_status.rb +33 -6
  69. data/lib/rails_error_dashboard/configuration.rb +39 -1
  70. data/lib/rails_error_dashboard/engine.rb +28 -0
  71. data/lib/rails_error_dashboard/manual_error_reporter.rb +16 -5
  72. data/lib/rails_error_dashboard/queries/analytics_stats.rb +89 -30
  73. data/lib/rails_error_dashboard/queries/baseline_stats.rb +107 -0
  74. data/lib/rails_error_dashboard/queries/dashboard_stats.rb +216 -74
  75. data/lib/rails_error_dashboard/queries/error_correlation.rb +6 -3
  76. data/lib/rails_error_dashboard/queries/errors_list.rb +15 -2
  77. data/lib/rails_error_dashboard/queries/event_volume.rb +503 -0
  78. data/lib/rails_error_dashboard/queries/similar_errors.rb +1 -1
  79. data/lib/rails_error_dashboard/services/analytics_cache_manager.rb +43 -18
  80. data/lib/rails_error_dashboard/services/backtrace_processor.rb +3 -1
  81. data/lib/rails_error_dashboard/services/breadcrumb_collector.rb +23 -0
  82. data/lib/rails_error_dashboard/services/cause_chain_extractor.rb +3 -1
  83. data/lib/rails_error_dashboard/services/codeberg_issue_client.rb +13 -2
  84. data/lib/rails_error_dashboard/services/diagnostic_dump_generator.rb +5 -3
  85. data/lib/rails_error_dashboard/services/encoding_sanitizer.rb +80 -0
  86. data/lib/rails_error_dashboard/services/error_broadcaster.rb +180 -32
  87. data/lib/rails_error_dashboard/services/error_hash_generator.rb +9 -3
  88. data/lib/rails_error_dashboard/services/error_notification_dispatcher.rb +13 -0
  89. data/lib/rails_error_dashboard/services/exception_filter.rb +55 -0
  90. data/lib/rails_error_dashboard/services/git_head_reader.rb +102 -0
  91. data/lib/rails_error_dashboard/services/notification_throttler.rb +172 -28
  92. data/lib/rails_error_dashboard/services/sensitive_data_filter.rb +47 -1
  93. data/lib/rails_error_dashboard/services/storm_protection/circuit_breaker.rb +59 -3
  94. data/lib/rails_error_dashboard/services/storm_protection/count_buffer.rb +64 -6
  95. data/lib/rails_error_dashboard/services/storm_protection/fingerprint_buckets.rb +25 -2
  96. data/lib/rails_error_dashboard/services/storm_protection/gate.rb +32 -5
  97. data/lib/rails_error_dashboard/services/swallowed_exception_tracker.rb +99 -23
  98. data/lib/rails_error_dashboard/services/url_safety.rb +40 -0
  99. data/lib/rails_error_dashboard/services/variable_serializer.rb +125 -10
  100. data/lib/rails_error_dashboard/subscribers/breadcrumb_subscriber.rb +111 -0
  101. data/lib/rails_error_dashboard/subscribers/issue_tracker_subscriber.rb +11 -2
  102. data/lib/rails_error_dashboard/value_objects/error_context.rb +40 -3
  103. data/lib/rails_error_dashboard/version.rb +1 -1
  104. data/lib/rails_error_dashboard.rb +34 -0
  105. data/lib/tasks/error_dashboard.rake +54 -4
  106. metadata +16 -2
@@ -0,0 +1,503 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RailsErrorDashboard
4
+ module Queries
5
+ # How many EVENTS happened inside a time window.
6
+ #
7
+ # The distinction this class exists to make: an ErrorLog row is a GROUP,
8
+ # and its occurred_at is FIRST-SEEN -- written once at creation and never
9
+ # rewritten when the error recurs. Filtering groups by occurred_at and
10
+ # summing occurrence_count therefore answers "how much lifetime volume do
11
+ # the groups born in this window carry?", which is not the same question as
12
+ # "how many events happened in this window?". An error first seen at 23:59
13
+ # that recurs at 00:01 reported zero errors today and two yesterday.
14
+ #
15
+ # An event lands in exactly one of three places, and the window total is
16
+ # the sum of all three:
17
+ #
18
+ # 1. an ErrorOccurrence row -- ordinary captures, with a real
19
+ # per-event timestamp
20
+ # 2. an EventCount bucket -- storm-shed events, which write no
21
+ # occurrence row by design; the bucket
22
+ # is their timestamp
23
+ # 3. the group's own count -- rows from before occurrence tracking
24
+ # existed, which have neither of the
25
+ # above. Counted against the group's
26
+ # occurred_at, which for such a row is
27
+ # the best (and only) timestamp there is
28
+ #
29
+ # Nothing is double counted: (1) and (2) are written by mutually exclusive
30
+ # paths, and (3) only ever covers the remainder of a group that has no
31
+ # per-event record at all.
32
+ class EventVolume
33
+ # @param scope [ActiveRecord::Relation] an ErrorLog scope (already
34
+ # filtered by application, if applicable)
35
+ # @param from [Time]
36
+ # @param to [Time, nil] exclusive upper bound; open-ended when nil
37
+ # @return [Integer]
38
+ def self.in_window(scope, from, to = nil)
39
+ new(scope, from, to).count
40
+ end
41
+
42
+ # Same window, bucketed by day: { Date => Integer }.
43
+ def self.by_day(scope, from, to = nil)
44
+ new(scope, from, to).by_day
45
+ end
46
+
47
+ # Same window, grouped by a column on the GROUP row (error_type,
48
+ # platform, environment): { value => Integer }.
49
+ #
50
+ # Every breakdown has to sum the same three terms as the headline total,
51
+ # or a page's parts stop adding up to its whole. Exposing one primitive
52
+ # is what stops each call site reimplementing that sum and drifting --
53
+ # which is precisely how the Analytics page came to disagree with the
54
+ # Overview.
55
+ def self.by_group_attribute(scope, column, from, to = nil)
56
+ new(scope, from, to).by_group_attribute(column)
57
+ end
58
+
59
+ def initialize(scope, from, to = nil)
60
+ @scope = scope
61
+ @from = from
62
+ @to = to
63
+ end
64
+
65
+ def count
66
+ occurrence_events + bucketed_events + untracked_events
67
+ end
68
+
69
+ def by_day
70
+ totals = Hash.new(0)
71
+ occurrence_events_by_day.each { |day, n| totals[day] += n }
72
+ bucketed_events_by_day.each { |day, n| totals[day] += n }
73
+ untracked_events_by_day.each { |day, n| totals[day] += n }
74
+ totals
75
+ end
76
+
77
+ # Events in the window, bucketed by hour-of-day (0..23) against each
78
+ # event's OWN timestamp -- the diurnal curve, not a time series.
79
+ # The hour is LOCAL for the same reason the day is: an operator asking
80
+ # when their errors peak means their own clock. Bucketed in Ruby from the
81
+ # windowed, already-aggregated rows so the offset used is the one in
82
+ # force at each instant.
83
+ def by_hour_of_day
84
+ zone = self.class.reporting_zone
85
+ totals = Hash.new(0)
86
+ (0..23).each { |h| totals[h] = 0 }
87
+
88
+ occurrence_events_by_hour.each { |ts, n| totals[local_hour(ts, zone)] += n }
89
+ bucketed_events_by_hour.each { |ts, n| totals[local_hour(ts, zone)] += n }
90
+ untracked_groups.each do |_id, remainder, occurred_at|
91
+ totals[occurred_at.in_time_zone(zone).hour] += remainder if occurred_at
92
+ end
93
+ totals
94
+ end
95
+
96
+ # Events in the window, grouped by a column on the ErrorLog row.
97
+ #
98
+ # All three terms are joined back to their group so they can be grouped
99
+ # by the group's own attribute: occurrence rows and buckets carry no
100
+ # error_type of their own. Aggregation stays in SQL (NFR-6) -- the only
101
+ # thing loaded into Ruby is the grouped result.
102
+ def by_group_attribute(column)
103
+ totals = Hash.new(0)
104
+ occurrence_events_by_attribute(column).each { |k, n| totals[k] += n }
105
+ bucketed_events_by_attribute(column).each { |k, n| totals[k] += n }
106
+ untracked_events_by_attribute(column).each { |k, n| totals[k] += n }
107
+ totals
108
+ end
109
+
110
+ private
111
+
112
+ # A SUBQUERY, not a plucked array of ids: the group set is unbounded and
113
+ # loading it into Ruby to pass back as an IN list is the shape that has
114
+ # caused unbounded-memory bugs in this codebase before. The database
115
+ # keeps the id set on its own side.
116
+ def group_ids
117
+ @group_ids ||= @scope.select(:id)
118
+ end
119
+
120
+ # (1) Ordinary captures.
121
+ def occurrence_events
122
+ return 0 unless occurrences_available?
123
+
124
+ window(ErrorOccurrence.where(error_log_id: group_ids), ErrorOccurrence.table_name).count
125
+ end
126
+
127
+ def occurrence_events_by_day
128
+ return {} unless occurrences_available?
129
+
130
+ rows = window(ErrorOccurrence.where(error_log_id: group_ids), ErrorOccurrence.table_name)
131
+ .group(day_expression(ErrorOccurrence.table_name, "occurred_at")).count
132
+
133
+ ruby_side_day_bucketing? ? group_by_local_day(rows) : rows.transform_keys { |k| to_date(k) }
134
+ end
135
+
136
+ # (2) Storm-shed events.
137
+ def bucketed_events
138
+ return 0 unless buckets_available?
139
+
140
+ window(EventCount.where(error_log_id: group_ids), EventCount.table_name, column: "bucket_at")
141
+ .sum(:count)
142
+ end
143
+
144
+ def bucketed_events_by_day
145
+ return {} unless buckets_available?
146
+
147
+ rows = window(EventCount.where(error_log_id: group_ids), EventCount.table_name, column: "bucket_at")
148
+ .group(day_expression(EventCount.table_name, "bucket_at")).sum(:count)
149
+
150
+ ruby_side_day_bucketing? ? group_by_local_day(rows) : rows.transform_keys { |k| to_date(k) }
151
+ end
152
+
153
+ # (3) Groups with no per-event record of any kind: rows created before
154
+ # occurrence tracking, and rows written directly. Their occurrence_count
155
+ # is the only evidence the events happened, and the group's own
156
+ # occurred_at is the only timestamp available for them.
157
+ def untracked_groups
158
+ @untracked_groups ||= begin
159
+ # Only groups whose own occurred_at falls in the window can
160
+ # contribute here at all, so the scan is bounded by the window rather
161
+ # than by the whole table.
162
+ #
163
+ # One SELECT with two correlated sub-selects, rather than three
164
+ # separate round trips: this runs on the capture path (the stats
165
+ # broadcast recomputes it) and the dashboard asks for several windows
166
+ # per render, so a per-window query count multiplies quickly.
167
+ rows = ErrorLog.connection.select_all(untracked_sql(window(@scope, ErrorLog.table_name)))
168
+ rows.filter_map do |row|
169
+ remainder = row["occurrence_count"].to_i - row["tracked"].to_i
170
+ next if remainder <= 0
171
+
172
+ [ row["id"], remainder, to_time(row["occurred_at"]) ]
173
+ end
174
+ end
175
+ end
176
+
177
+ def untracked_sql(candidates)
178
+ logs = ErrorLog.table_name
179
+ occurrence_term =
180
+ if occurrences_available?
181
+ "(SELECT COUNT(*) FROM #{ErrorOccurrence.table_name} o " \
182
+ "WHERE o.error_log_id = #{logs}.id)"
183
+ else
184
+ "0"
185
+ end
186
+ bucket_term =
187
+ if buckets_available?
188
+ "(SELECT COALESCE(SUM(b.count), 0) FROM #{EventCount.table_name} b " \
189
+ "WHERE b.error_log_id = #{logs}.id)"
190
+ else
191
+ "0"
192
+ end
193
+
194
+ candidates
195
+ .select(Arel.sql("#{logs}.id, #{logs}.occurrence_count, #{logs}.occurred_at, " \
196
+ "#{occurrence_term} + #{bucket_term} AS tracked"))
197
+ .to_sql
198
+ end
199
+
200
+ def to_time(value)
201
+ return value if value.respond_to?(:to_date) && !value.is_a?(String)
202
+
203
+ Time.zone ? Time.zone.parse(value.to_s) : Time.parse(value.to_s)
204
+ rescue StandardError
205
+ nil
206
+ end
207
+
208
+ # Grouped by the HOUR, never by the raw timestamp.
209
+ #
210
+ # Grouping by the raw timestamp returned one row per distinct instant:
211
+ # 1,000 events inside one second produced 1,000 intermediate entries in
212
+ # Ruby to compute 24 bins. Memory has to be bounded by the reporting
213
+ # WINDOW, not by how many distinct timestamps happen to be in it -- a
214
+ # burst is exactly when these numbers matter and exactly when the row
215
+ # count explodes.
216
+ #
217
+ # On PostgreSQL/MySQL the local hour is derived in SQL, so at most 24
218
+ # rows come back. SQLite has no tz database, so it truncates to a
219
+ # FIXED-WIDTH UTC bin in SQL (bounding the result by the window) and Ruby
220
+ # converts that bin's instant to the local hour.
221
+ def occurrence_events_by_hour
222
+ return {} unless occurrences_available?
223
+
224
+ table = ErrorOccurrence.table_name
225
+ window(ErrorOccurrence.where(error_log_id: group_ids), table)
226
+ .group(hour_expression(table, "occurred_at")).count
227
+ end
228
+
229
+ def bucketed_events_by_hour
230
+ return {} unless buckets_available?
231
+
232
+ table = EventCount.table_name
233
+ window(EventCount.where(error_log_id: group_ids), table, column: "bucket_at")
234
+ .group(hour_expression(table, "bucket_at")).sum(:count)
235
+ end
236
+
237
+ # The width of the SQLite grouping bin, in seconds.
238
+ #
239
+ # A local hour boundary must always fall on a bin EDGE, or two events on
240
+ # opposite sides of it collapse into one bin and can no longer be told
241
+ # apart: in Asia/Kolkata (+05:30) 00:15 and 00:45 UTC are local hours 5
242
+ # and 6, but share a UTC hour. So the bin has to divide every zone offset
243
+ # in use. Offsets are whole multiples of 15 minutes (+05:30, +05:45,
244
+ # -09:30 and the rest), which makes 15 minutes the widest safe bin -- the
245
+ # same 900s quantum EventCount::BUCKET_SECONDS uses, for the same
246
+ # divides-the-clock-cleanly reason.
247
+ #
248
+ # Width matters only for the row count, which stays bounded by the
249
+ # WINDOW (4 rows per hour) rather than by the number of distinct event
250
+ # timestamps in it.
251
+ HOUR_BIN_SECONDS = 900
252
+
253
+ # The grouping key for hour-of-day aggregation.
254
+ #
255
+ # Returns the local hour directly where the adapter can convert zones,
256
+ # and a UTC bin key where it cannot. local_hour handles both: a Numeric
257
+ # passes straight through, a key is parsed AS UTC and converted.
258
+ def hour_expression(table, column)
259
+ zone = self.class.reporting_zone
260
+ quoted = ErrorLog.connection.quote(zone.tzinfo.name)
261
+
262
+ Arel.sql(
263
+ case ErrorLog.connection.adapter_name.downcase
264
+ when /postgres/
265
+ "EXTRACT(HOUR FROM #{table}.#{column} AT TIME ZONE 'UTC' AT TIME ZONE #{quoted})"
266
+ when /mysql|trilogy/
267
+ # Named zone, not a numeric offset -- same DST reasoning as
268
+ # day_expression.
269
+ "HOUR(CONVERT_TZ(#{table}.#{column}, '+00:00', #{quoted}))"
270
+ else
271
+ # SQLite: bound the row count by truncating the UTC epoch second to
272
+ # a whole bin. Integer division floors, which is what keeps every
273
+ # bin edge on a multiple of HOUR_BIN_SECONDS from the epoch -- and
274
+ # therefore on every local hour boundary. Ruby then shifts the bin
275
+ # into the reporting zone.
276
+ "(CAST(strftime('%s', #{table}.#{column}) AS INTEGER) / #{HOUR_BIN_SECONDS}) * #{HOUR_BIN_SECONDS}"
277
+ end
278
+ )
279
+ end
280
+
281
+ # The adapter may hand back either an hour already binned in SQL
282
+ # (PostgreSQL/MySQL, as a Numeric or a numeric string) or a UTC bin key
283
+ # that still needs converting (SQLite: epoch seconds).
284
+ #
285
+ # SQLite's key is an epoch second, so it carries its own UTC meaning and
286
+ # there is nothing to misread. The earlier key was a bare
287
+ # 'YYYY-MM-DD HH:00:00' string, which Time.zone.parse read as LOCAL time
288
+ # -- reporting the UTC hour verbatim.
289
+ def local_hour(value, zone)
290
+ return Time.at(value.to_i).utc.in_time_zone(zone).hour if sqlite_hour_bins?
291
+
292
+ return value.to_i % 24 if value.is_a?(Numeric)
293
+ return value.to_i % 24 if value.is_a?(String) && value.match?(/\A\d+(\.\d+)?\z/)
294
+
295
+ time = to_time(value)
296
+ time ? time.in_time_zone(zone).hour : 0
297
+ end
298
+
299
+ # True when hour_expression fell through to the SQLite branch and the
300
+ # keys are UTC bin epochs rather than hours binned in SQL.
301
+ def sqlite_hour_bins?
302
+ !ErrorLog.connection.adapter_name.downcase.match?(/postgres|mysql|trilogy/)
303
+ end
304
+
305
+ # The three by-attribute terms. Each joins back to ErrorLog because the
306
+ # attribute being grouped by lives on the GROUP, not on the event row.
307
+ def occurrence_events_by_attribute(column)
308
+ return {} unless occurrences_available?
309
+
310
+ logs = ErrorLog.table_name
311
+ window(
312
+ ErrorOccurrence.where(error_log_id: group_ids)
313
+ .joins("INNER JOIN #{logs} ON #{logs}.id = #{ErrorOccurrence.table_name}.error_log_id"),
314
+ ErrorOccurrence.table_name
315
+ ).group("#{logs}.#{column}").count
316
+ end
317
+
318
+ def bucketed_events_by_attribute(column)
319
+ return {} unless buckets_available?
320
+
321
+ logs = ErrorLog.table_name
322
+ window(
323
+ EventCount.where(error_log_id: group_ids)
324
+ .joins("INNER JOIN #{logs} ON #{logs}.id = #{EventCount.table_name}.error_log_id"),
325
+ EventCount.table_name,
326
+ column: "bucket_at"
327
+ ).group("#{logs}.#{column}").sum(:count)
328
+ end
329
+
330
+ # The untracked remainder is already resolved to (id, remainder), so the
331
+ # attribute is fetched for just those ids -- a bounded set, since only
332
+ # groups whose own occurred_at falls in the window can contribute.
333
+ def untracked_events_by_attribute(column)
334
+ rows = untracked_groups
335
+ return {} if rows.empty?
336
+
337
+ attributes = ErrorLog.where(id: rows.map(&:first)).pluck(:id, column).to_h
338
+ totals = Hash.new(0)
339
+ rows.each { |id, remainder, _occurred_at| totals[attributes[id]] += remainder }
340
+ totals
341
+ end
342
+
343
+ def untracked_events
344
+ untracked_groups.sum { |_id, remainder, _occurred_at| remainder }
345
+ end
346
+
347
+ # Already in Ruby, so the zone conversion is direct -- and uses the
348
+ # offset in force at each row's own instant.
349
+ def untracked_events_by_day
350
+ zone = self.class.reporting_zone
351
+ totals = Hash.new(0)
352
+ untracked_groups.each do |_id, remainder, occurred_at|
353
+ next unless occurred_at
354
+
355
+ totals[occurred_at.in_time_zone(zone).to_date] += remainder
356
+ end
357
+ totals
358
+ end
359
+
360
+ def window(relation, table, column: "occurred_at")
361
+ relation = relation.where("#{table}.#{column} >= ?", @from)
362
+ @to ? relation.where("#{table}.#{column} < ?", @to) : relation
363
+ end
364
+
365
+ # Grouping by day has to happen in SQL -- loading rows to bucket them in
366
+ # Ruby is exactly the unbounded-memory shape this codebase avoids.
367
+ # The reporting zone: one definition, used for BOTH the SQL bucket key
368
+ # and the Ruby-side window boundaries, so the two cannot disagree.
369
+ def self.reporting_zone
370
+ Time.zone || ActiveSupport::TimeZone["UTC"]
371
+ end
372
+
373
+ # Group by calendar day IN THE APPLICATION TIME ZONE.
374
+ #
375
+ # Timestamps are stored in UTC. A bare DATE(column) therefore buckets by
376
+ # UTC day while the caller looks up Date.current in Time.zone -- at 00:15
377
+ # in Asia/Kolkata a fresh capture is stored as 18:45 the previous day UTC,
378
+ # so "today" reported zero. The offset must also be the one in force AT
379
+ # EACH ROW'S OWN TIMESTAMP, not one current offset applied to the whole
380
+ # window, or a window spanning a DST change misplaces every row on one
381
+ # side of it.
382
+ #
383
+ # PostgreSQL and MySQL have a tz database and do this per row natively.
384
+ # SQLite has neither AT TIME ZONE nor CONVERT_TZ, and its 'localtime'
385
+ # modifier uses the SERVER's zone rather than the application's -- so
386
+ # there the conversion is done in Ruby, where the zone object knows each
387
+ # instant's true offset. That path is bounded: it groups an already
388
+ # windowed relation, and only the grouped result reaches Ruby.
389
+ def day_expression(table, column)
390
+ zone = self.class.reporting_zone
391
+
392
+ Arel.sql(
393
+ case ErrorLog.connection.adapter_name.downcase
394
+ when /postgres/
395
+ "DATE(#{table}.#{column} AT TIME ZONE 'UTC' AT TIME ZONE #{ErrorLog.connection.quote(zone.tzinfo.name)})"
396
+ when /mysql|trilogy/
397
+ # A NAMED zone, not a numeric offset. CONVERT_TZ with '+05:30'
398
+ # applies one fixed offset to every row, which silently misplaces
399
+ # rows on the far side of a DST transition; the named form uses the
400
+ # offset in force at each row's own timestamp.
401
+ #
402
+ # This needs the server's time-zone tables (mysql_tzinfo_to_sql) --
403
+ # the same requirement groupdate already imposes for every chart on
404
+ # the dashboard, and which `rails error_dashboard:verify` checks.
405
+ # See docs/guides/DATABASE_OPTIONS.md.
406
+ "DATE(CONVERT_TZ(#{table}.#{column}, '+00:00', #{ErrorLog.connection.quote(zone.tzinfo.name)}))"
407
+ else
408
+ # SQLite: no tz database, so the fold into local days happens in
409
+ # Ruby (see group_by_local_day). The SQL key must still be BOUNDED
410
+ # -- grouping by the raw timestamp returned one row per distinct
411
+ # instant, so a burst of 200 events in one day handed Ruby 200 rows
412
+ # to produce a single daily total. Memory has to be bounded by the
413
+ # reporting WINDOW, not by how many distinct timestamps are in it.
414
+ #
415
+ # The bound is a 15-minute truncated UTC bin, the same width the
416
+ # storm rollup uses and for the same reason: 15 minutes divides
417
+ # every UTC offset in use (including +05:30 Kolkata and +05:45
418
+ # Kathmandu), so a LOCAL day boundary always falls on a bin edge
419
+ # and no bin ever straddles two local days. An hour-wide bin would
420
+ # NOT be safe in those zones.
421
+ #
422
+ # The key is the bin's EPOCH SECOND, not a datetime string. An
423
+ # epoch second carries its own UTC meaning, so there is nothing to
424
+ # misread; a bare 'YYYY-MM-DD HH:MM:SS' string is parsed as LOCAL
425
+ # by Time.zone.parse, which would shift every bin by the reporting
426
+ # zone's offset. Integer division floors, keeping every bin edge on
427
+ # a multiple of the width from the epoch -- and therefore on every
428
+ # local midnight.
429
+ "(CAST(strftime('%s', #{table}.#{column}) AS INTEGER) / #{DAY_BIN_SECONDS}) * #{DAY_BIN_SECONDS}"
430
+ end
431
+ )
432
+ end
433
+
434
+ # The width of the SQLite day-grouping bin, in seconds -- the same 900s
435
+ # quantum, and the same reasoning, as HOUR_BIN_SECONDS above: it has to
436
+ # divide every zone offset in use so a local DAY boundary falls on a bin
437
+ # EDGE and no bin ever straddles two local days.
438
+ #
439
+ # This must equal EventCount::BUCKET_SECONDS, and there is a spec that
440
+ # asserts it. It is a literal rather than a reference because EventCount
441
+ # is an autoloaded model and this constant is evaluated at load time.
442
+ DAY_BIN_SECONDS = 900
443
+
444
+ # True when the adapter cannot convert zones itself and Ruby must.
445
+ def ruby_side_day_bucketing?
446
+ !ErrorLog.connection.adapter_name.downcase.match?(/postgres|mysql|trilogy/)
447
+ end
448
+
449
+ # Collapse a { utc instant => count } result into { Date => count } using
450
+ # the zone's offset AT EACH instant -- which is what makes a DST-spanning
451
+ # window correct. The keys are 15-minute bins (see day_expression), and
452
+ # because that width divides every offset in use, a bin never straddles
453
+ # two local days: folding by the bin's own instant is exact.
454
+ def group_by_local_day(rows)
455
+ zone = self.class.reporting_zone
456
+ totals = Hash.new(0)
457
+ rows.each do |key, value|
458
+ time = to_utc_bin_time(key)
459
+ next unless time
460
+
461
+ totals[time.in_time_zone(zone).to_date] += value
462
+ end
463
+ totals
464
+ end
465
+
466
+ # day_expression's SQLite key is a UTC epoch second (see there), which is
467
+ # unambiguous. Anything else reaching here is already a Time-like value
468
+ # from another adapter, so it is used as-is -- deliberately NOT routed
469
+ # through Time.zone.parse, which reads a bare datetime string as LOCAL.
470
+ def to_utc_bin_time(value)
471
+ return Time.at(value.to_i).utc if value.is_a?(Numeric)
472
+ return Time.at(value.to_i).utc if value.is_a?(String) && value.match?(/\A-?\d+\z/)
473
+ return value if value.respond_to?(:in_time_zone) && !value.is_a?(String)
474
+
475
+ nil
476
+ end
477
+
478
+ def to_date(value)
479
+ return value if value.is_a?(Date)
480
+
481
+ value.respond_to?(:to_date) ? value.to_date : Date.parse(value.to_s)
482
+ rescue StandardError
483
+ value
484
+ end
485
+
486
+ def occurrences_available?
487
+ return @occurrences_available if defined?(@occurrences_available)
488
+
489
+ @occurrences_available = defined?(ErrorOccurrence) && ErrorOccurrence.table_exists?
490
+ rescue StandardError
491
+ @occurrences_available = false
492
+ end
493
+
494
+ def buckets_available?
495
+ return @buckets_available if defined?(@buckets_available)
496
+
497
+ @buckets_available = defined?(EventCount) && EventCount.table_exists?
498
+ rescue StandardError
499
+ @buckets_available = false
500
+ end
501
+ end
502
+ end
503
+ end
@@ -78,7 +78,7 @@ module RailsErrorDashboard
78
78
  if error_prefix.present?
79
79
  candidates += ErrorLog
80
80
  .where(platform: target_error.platform)
81
- .where("error_type LIKE ?", "%#{error_prefix}%")
81
+ .where("error_type LIKE ? ESCAPE '!'", "%#{ActiveRecord::Base.sanitize_sql_like(error_prefix, "!")}%")
82
82
  .where.not(id: target_error.id)
83
83
  .where.not(id: candidates.map(&:id))
84
84
  .limit(20)
@@ -2,29 +2,54 @@
2
2
 
3
3
  module RailsErrorDashboard
4
4
  module Services
5
- # Infrastructure service: Clear analytics caches
5
+ # Infrastructure service: invalidate the dashboard's cached statistics
6
6
  #
7
- # Clears dashboard_stats, analytics_stats, and platform_comparison
8
- # cache entries. Handles cache stores that don't support delete_matched
9
- # (e.g., SolidCache) gracefully.
7
+ # Every cached stats key (DashboardStats, AnalyticsStats) embeds a
8
+ # GENERATION number read from Rails.cache. Invalidation is one increment of
9
+ # that number: entries written under the old generation are simply never
10
+ # read again and age out by their own TTL.
11
+ #
12
+ # This replaced delete_matched("dashboard_stats/*") and friends, which is a
13
+ # SCAN of the HOST app's whole Redis keyspace, NotImplementedError on
14
+ # memcached, and used to run from ErrorLog's after_save -- i.e. inside the
15
+ # host's request thread on every captured error.
16
+ #
17
+ # Who calls .clear: user actions and maintenance that change what the cards
18
+ # show (resolve, status change, batch actions, mute, retention). Captures
19
+ # deliberately do NOT: they rely on the TTL (1 minute for the stat cards,
20
+ # 5 for analytics), which decouples capture cost from dashboard freshness.
21
+ #
22
+ # With :memory_store the generation is per process, so an action handled by
23
+ # one worker reaches the others when their entries expire. With :null_store
24
+ # nothing is cached and there is nothing to invalidate.
10
25
  class AnalyticsCacheManager
11
- CACHE_PATTERNS = %w[
12
- dashboard_stats/*
13
- analytics_stats/*
14
- platform_comparison/*
15
- ].freeze
26
+ GENERATION_KEY = "red/cache_gen"
27
+
28
+ # @return [Integer] the current cache generation; 0 when unset or unreadable
29
+ def self.generation
30
+ Rails.cache.read(GENERATION_KEY).to_i
31
+ rescue => e
32
+ RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] cache generation unreadable: #{e.class}: #{e.message}")
33
+ 0
34
+ end
16
35
 
17
- # Clear all analytics caches
36
+ # Invalidate every cached dashboard statistic. Never raises.
37
+ #
38
+ # A plain read + write, deliberately not Rails.cache.increment: increment
39
+ # is unimplemented on some stores, returns nil for a missing key on
40
+ # others, and on Redis is an INCRBY that fails against an entry written
41
+ # by #write. It does not need to be atomic either -- racing clears only
42
+ # need the number to differ from the one stale entries were written under.
43
+ #
44
+ # Never lower than the clock in milliseconds, so that a store which evicts
45
+ # the generation key cannot restart at a number that entries still inside
46
+ # their TTL were written under a few minutes earlier.
18
47
  def self.clear
19
- if Rails.cache.respond_to?(:delete_matched)
20
- CACHE_PATTERNS.each { |pattern| Rails.cache.delete_matched(pattern) }
21
- else
22
- Rails.logger.info("Cache store doesn't support delete_matched, skipping cache clear") if Rails.logger
23
- end
24
- rescue NotImplementedError => e
25
- Rails.logger.info("Cache store doesn't support delete_matched: #{e.message}") if Rails.logger
48
+ Rails.cache.write(GENERATION_KEY, [ generation + 1, (Time.now.to_f * 1000).to_i ].max)
49
+ nil
26
50
  rescue => e
27
- Rails.logger.error("Failed to clear analytics cache: #{e.message}") if Rails.logger
51
+ RailsErrorDashboard::Logger.error("[RailsErrorDashboard] Failed to clear analytics cache: #{e.class}: #{e.message}")
52
+ nil
28
53
  end
29
54
  end
30
55
  end
@@ -22,7 +22,9 @@ module RailsErrorDashboard
22
22
 
23
23
  max_lines ||= RailsErrorDashboard.configuration.max_backtrace_lines
24
24
 
25
- limited_backtrace = backtrace.first(max_lines).map { |line| shorten_gem_path(line) }
25
+ # Scrubbed per line: a frame label can hold any bytes, and both the sub
26
+ # below and the join (mixed encodings) raise on an invalid one.
27
+ limited_backtrace = backtrace.first(max_lines).map { |line| shorten_gem_path(EncodingSanitizer.scrub(line)) }
26
28
  result = limited_backtrace.join("\n")
27
29
 
28
30
  if backtrace.length > max_lines
@@ -77,6 +77,29 @@ module RailsErrorDashboard
77
77
  nil
78
78
  end
79
79
 
80
+ # Open a buffer ONLY if this thread has none, and say whether we opened
81
+ # it. The caller passes that answer back to clear_buffer_if_owned, so a
82
+ # job performed inline inside a request adds its crumbs to the request's
83
+ # trail and does not tear it down on the way out.
84
+ #
85
+ # Unconditional init/clear here would erase a surrounding request's
86
+ # buffer on every perform_now, the test adapter and the :inline queue.
87
+ # @return [Boolean] true when this caller opened the buffer
88
+ def self.init_buffer_unless_present
89
+ return false if Thread.current[THREAD_KEY]
90
+
91
+ init_buffer
92
+ true
93
+ rescue => e
94
+ RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] BreadcrumbCollector.init_buffer_unless_present failed: #{e.message}")
95
+ false
96
+ end
97
+
98
+ # Tear down only what this caller opened (see init_buffer_unless_present).
99
+ def self.clear_buffer_if_owned(owned)
100
+ clear_buffer if owned
101
+ end
102
+
80
103
  # Clear the ring buffer (end of request — MUST be called in ensure block)
81
104
  def self.clear_buffer
82
105
  Thread.current[THREAD_KEY] = nil
@@ -43,7 +43,9 @@ module RailsErrorDashboard
43
43
  depth += 1
44
44
  end
45
45
 
46
- chain.empty? ? nil : chain.to_json
46
+ # to_json raises on a cause message or frame with invalid bytes, which
47
+ # used to drop the whole chain.
48
+ chain.empty? ? nil : EncodingSanitizer.scrub_deep(chain).to_json
47
49
  rescue => e
48
50
  # SAFETY: Never let cause chain extraction break error logging
49
51
  RailsErrorDashboard::Logger.debug(
@@ -40,7 +40,7 @@ module RailsErrorDashboard
40
40
  auth_headers
41
41
  )
42
42
 
43
- response[:status] == 201 ? success_response({}) : error_response("Codeberg API error (#{response[:status]})")
43
+ patch_result(response)
44
44
  end
45
45
 
46
46
  def reopen_issue(number:)
@@ -50,7 +50,7 @@ module RailsErrorDashboard
50
50
  auth_headers
51
51
  )
52
52
 
53
- response[:status] == 201 ? success_response({}) : error_response("Codeberg API error (#{response[:status]})")
53
+ patch_result(response)
54
54
  end
55
55
 
56
56
  def add_comment(number:, body:)
@@ -114,6 +114,17 @@ module RailsErrorDashboard
114
114
 
115
115
  private
116
116
 
117
+ # Gitea/Forgejo answer a successful PATCH with 200; 201 is what they
118
+ # return for creation. Accepting only 201 reported every close and reopen
119
+ # as failed although the forge had applied it.
120
+ def patch_result(response)
121
+ if [ 200, 201 ].include?(response[:status])
122
+ success_response({})
123
+ else
124
+ error_response("Codeberg API error (#{response[:status]})")
125
+ end
126
+ end
127
+
117
128
  def auth_headers
118
129
  { "Authorization" => "token #{@token}" }
119
130
  end