rails_error_dashboard 0.12.1 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/app/controllers/rails_error_dashboard/application_controller.rb +92 -11
- data/app/controllers/rails_error_dashboard/errors_controller.rb +111 -34
- data/app/controllers/rails_error_dashboard/webhooks_controller.rb +3 -0
- data/app/helpers/rails_error_dashboard/application_helper.rb +46 -1
- data/app/helpers/rails_error_dashboard/backtrace_helper.rb +8 -0
- data/app/jobs/rails_error_dashboard/async_error_logging_job.rb +10 -0
- data/app/jobs/rails_error_dashboard/concerns/plain_channel_message.rb +71 -0
- data/app/jobs/rails_error_dashboard/notification_burst_summary_job.rb +55 -0
- data/app/jobs/rails_error_dashboard/retention_cleanup_job.rb +122 -1
- data/app/jobs/rails_error_dashboard/storm_flush_job.rb +7 -4
- data/app/jobs/rails_error_dashboard/storm_notification_job.rb +8 -44
- data/app/models/rails_error_dashboard/error_baseline.rb +12 -9
- data/app/models/rails_error_dashboard/error_comment.rb +0 -5
- data/app/models/rails_error_dashboard/error_log.rb +19 -3
- data/app/models/rails_error_dashboard/error_logs_record.rb +34 -0
- data/app/models/rails_error_dashboard/error_occurrence.rb +9 -1
- data/app/models/rails_error_dashboard/event_count.rb +132 -0
- data/app/models/rails_error_dashboard/event_timing_gap.rb +55 -0
- data/app/views/layouts/rails_error_dashboard.html.erb +58 -5
- data/app/views/rails_error_dashboard/errors/_discussion.html.erb +18 -11
- data/app/views/rails_error_dashboard/errors/_issue_section.html.erb +22 -8
- data/app/views/rails_error_dashboard/errors/_request_context.html.erb +2 -0
- data/app/views/rails_error_dashboard/errors/_sidebar_metadata.html.erb +15 -6
- data/app/views/rails_error_dashboard/errors/analytics.html.erb +8 -8
- data/app/views/rails_error_dashboard/errors/diagnostic_dumps.html.erb +1 -1
- data/app/views/rails_error_dashboard/errors/index.html.erb +6 -1
- data/app/views/rails_error_dashboard/errors/overview.html.erb +12 -0
- data/app/views/rails_error_dashboard/errors/platform_comparison.html.erb +7 -7
- data/app/views/rails_error_dashboard/errors/releases.html.erb +2 -2
- data/app/views/rails_error_dashboard/errors/settings.html.erb +2 -0
- data/app/views/rails_error_dashboard/errors/show.html.erb +3 -3
- data/config/locales/de.yml +30 -0
- data/config/locales/en.yml +37 -0
- data/config/locales/es.yml +30 -0
- data/config/locales/fr.yml +30 -0
- data/config/locales/it.yml +30 -0
- data/config/locales/ja.yml +30 -0
- data/config/locales/pl.yml +30 -0
- data/config/locales/pt-BR.yml +30 -0
- data/config/locales/ru.yml +30 -0
- data/config/locales/uk.yml +30 -0
- data/config/locales/zh-CN.yml +30 -0
- data/db/migrate/20260917000001_add_last_notified_at_to_error_logs.rb +40 -0
- data/db/migrate/20260919000001_create_event_counts.rb +71 -0
- data/db/migrate/20260920000001_add_buckets_incomplete_to_storm_events.rb +25 -0
- data/db/migrate/20260920000002_create_event_timing_gaps.rb +55 -0
- data/lib/generators/rails_error_dashboard/install/templates/initializer.rb +12 -1
- data/lib/rails_error_dashboard/commands/assign_error.rb +12 -2
- data/lib/rails_error_dashboard/commands/backfill_environments.rb +2 -0
- data/lib/rails_error_dashboard/commands/backfill_resolved_at.rb +42 -0
- data/lib/rails_error_dashboard/commands/batch_delete_errors.rb +1 -0
- data/lib/rails_error_dashboard/commands/batch_mute_errors.rb +2 -0
- data/lib/rails_error_dashboard/commands/batch_resolve_errors.rb +2 -0
- data/lib/rails_error_dashboard/commands/batch_unmute_errors.rb +2 -0
- data/lib/rails_error_dashboard/commands/find_or_increment_error.rb +109 -11
- data/lib/rails_error_dashboard/commands/flush_rack_attack_events.rb +3 -1
- data/lib/rails_error_dashboard/commands/flush_storm_counts.rb +252 -12
- data/lib/rails_error_dashboard/commands/flush_swallowed_exceptions.rb +5 -2
- data/lib/rails_error_dashboard/commands/link_existing_issue.rb +1 -0
- data/lib/rails_error_dashboard/commands/log_error.rb +291 -40
- data/lib/rails_error_dashboard/commands/mute_error.rb +1 -0
- data/lib/rails_error_dashboard/commands/resolve_error.rb +2 -0
- data/lib/rails_error_dashboard/commands/scrub_invalid_encoding.rb +104 -0
- data/lib/rails_error_dashboard/commands/snooze_error.rb +34 -10
- data/lib/rails_error_dashboard/commands/unmute_error.rb +1 -0
- data/lib/rails_error_dashboard/commands/update_error_priority.rb +23 -2
- data/lib/rails_error_dashboard/commands/update_error_status.rb +33 -6
- data/lib/rails_error_dashboard/configuration.rb +39 -1
- data/lib/rails_error_dashboard/engine.rb +28 -0
- data/lib/rails_error_dashboard/manual_error_reporter.rb +16 -5
- data/lib/rails_error_dashboard/queries/analytics_stats.rb +89 -30
- data/lib/rails_error_dashboard/queries/baseline_stats.rb +107 -0
- data/lib/rails_error_dashboard/queries/dashboard_stats.rb +216 -74
- data/lib/rails_error_dashboard/queries/error_correlation.rb +6 -3
- data/lib/rails_error_dashboard/queries/errors_list.rb +15 -2
- data/lib/rails_error_dashboard/queries/event_volume.rb +503 -0
- data/lib/rails_error_dashboard/queries/similar_errors.rb +1 -1
- data/lib/rails_error_dashboard/services/analytics_cache_manager.rb +43 -18
- data/lib/rails_error_dashboard/services/backtrace_processor.rb +3 -1
- data/lib/rails_error_dashboard/services/breadcrumb_collector.rb +23 -0
- data/lib/rails_error_dashboard/services/cause_chain_extractor.rb +3 -1
- data/lib/rails_error_dashboard/services/codeberg_issue_client.rb +13 -2
- data/lib/rails_error_dashboard/services/diagnostic_dump_generator.rb +5 -3
- data/lib/rails_error_dashboard/services/encoding_sanitizer.rb +80 -0
- data/lib/rails_error_dashboard/services/error_broadcaster.rb +180 -32
- data/lib/rails_error_dashboard/services/error_hash_generator.rb +9 -3
- data/lib/rails_error_dashboard/services/error_notification_dispatcher.rb +13 -0
- data/lib/rails_error_dashboard/services/exception_filter.rb +55 -0
- data/lib/rails_error_dashboard/services/git_head_reader.rb +102 -0
- data/lib/rails_error_dashboard/services/notification_throttler.rb +172 -28
- data/lib/rails_error_dashboard/services/sensitive_data_filter.rb +47 -1
- data/lib/rails_error_dashboard/services/storm_protection/circuit_breaker.rb +59 -3
- data/lib/rails_error_dashboard/services/storm_protection/count_buffer.rb +64 -6
- data/lib/rails_error_dashboard/services/storm_protection/fingerprint_buckets.rb +25 -2
- data/lib/rails_error_dashboard/services/storm_protection/gate.rb +32 -5
- data/lib/rails_error_dashboard/services/swallowed_exception_tracker.rb +99 -23
- data/lib/rails_error_dashboard/services/url_safety.rb +40 -0
- data/lib/rails_error_dashboard/services/variable_serializer.rb +125 -10
- data/lib/rails_error_dashboard/subscribers/breadcrumb_subscriber.rb +111 -0
- data/lib/rails_error_dashboard/subscribers/issue_tracker_subscriber.rb +11 -2
- data/lib/rails_error_dashboard/value_objects/error_context.rb +40 -3
- data/lib/rails_error_dashboard/version.rb +1 -1
- data/lib/rails_error_dashboard.rb +34 -0
- data/lib/tasks/error_dashboard.rake +54 -4
- metadata +16 -2
|
@@ -5,6 +5,13 @@ module RailsErrorDashboard
|
|
|
5
5
|
# Query: Fetch dashboard statistics
|
|
6
6
|
# This is a read operation that aggregates error data for the dashboard
|
|
7
7
|
class DashboardStats
|
|
8
|
+
# The widest window any figure on the Overview covers (total_month).
|
|
9
|
+
# The completeness predicate is matched to this, not to "today".
|
|
10
|
+
WIDEST_DISPLAYED_WINDOW = 30.days
|
|
11
|
+
|
|
12
|
+
# One instance answers ONE call: several aggregates (today's event count,
|
|
13
|
+
# the 7-day trend, spike detection) are memoised on it so that they are
|
|
14
|
+
# computed once per call rather than once per card that shows them.
|
|
8
15
|
def initialize(application_id: nil)
|
|
9
16
|
@application_id = application_id
|
|
10
17
|
end
|
|
@@ -24,9 +31,9 @@ module RailsErrorDashboard
|
|
|
24
31
|
# rows reported five users hitting one error as "1 error today".
|
|
25
32
|
# Summing is also exact during a storm: counted-only events never
|
|
26
33
|
# create occurrence rows, but they DO raise occurrence_count.
|
|
27
|
-
total_today:
|
|
28
|
-
total_week:
|
|
29
|
-
total_month:
|
|
34
|
+
total_today: today_event_count,
|
|
35
|
+
total_week: week_event_counts.values.sum,
|
|
36
|
+
total_month: month_event_count,
|
|
30
37
|
unresolved: base_scope.unresolved.count,
|
|
31
38
|
resolved: base_scope.resolved.count,
|
|
32
39
|
reopened: reopened_count,
|
|
@@ -51,6 +58,13 @@ module RailsErrorDashboard
|
|
|
51
58
|
# window, the dimension is incomplete and the page says so
|
|
52
59
|
# rather than presenting an undercount as fact.
|
|
53
60
|
affected_users_incomplete: affected_users_incomplete?,
|
|
61
|
+
# A storm episode in this window lost its per-bucket TIMING
|
|
62
|
+
# evidence, so every time-window figure for it rests on the
|
|
63
|
+
# group's own occurred_at rather than on per-event records.
|
|
64
|
+
# Reported for the same reason as affected_users_incomplete:
|
|
65
|
+
# an incomplete dimension should say so rather than present a
|
|
66
|
+
# figure of unknown completeness as fact.
|
|
67
|
+
event_timing_incomplete: event_timing_incomplete?,
|
|
54
68
|
data_unavailable: false
|
|
55
69
|
}
|
|
56
70
|
end
|
|
@@ -68,6 +82,7 @@ module RailsErrorDashboard
|
|
|
68
82
|
{
|
|
69
83
|
data_unavailable: true,
|
|
70
84
|
affected_users_incomplete: false,
|
|
85
|
+
event_timing_incomplete: false,
|
|
71
86
|
total_today: 0,
|
|
72
87
|
total_week: 0,
|
|
73
88
|
total_month: 0,
|
|
@@ -93,13 +108,14 @@ module RailsErrorDashboard
|
|
|
93
108
|
end
|
|
94
109
|
|
|
95
110
|
def cache_key
|
|
96
|
-
#
|
|
97
|
-
#
|
|
98
|
-
#
|
|
111
|
+
# The cache GENERATION, not maximum(:updated_at): the timestamp cost a
|
|
112
|
+
# query per key build and changed on every capture, so the cache never
|
|
113
|
+
# hit while errors were arriving. Freshness after a capture is the
|
|
114
|
+
# 1-minute TTL; user actions bump the generation (AnalyticsCacheManager).
|
|
99
115
|
[
|
|
100
116
|
"dashboard_stats",
|
|
101
117
|
@application_id || "all",
|
|
102
|
-
|
|
118
|
+
Services::AnalyticsCacheManager.generation,
|
|
103
119
|
Time.current.hour
|
|
104
120
|
].join("/")
|
|
105
121
|
end
|
|
@@ -112,14 +128,29 @@ module RailsErrorDashboard
|
|
|
112
128
|
scope
|
|
113
129
|
end
|
|
114
130
|
|
|
115
|
-
# Total EVENTS in a window
|
|
116
|
-
#
|
|
131
|
+
# Total EVENTS in a window -- the events that HAPPENED in it, not the
|
|
132
|
+
# lifetime volume of the groups first seen in it.
|
|
133
|
+
#
|
|
134
|
+
# This used to filter groups by occurred_at and sum occurrence_count.
|
|
135
|
+
# occurred_at is first-seen and is never rewritten on recurrence, so an
|
|
136
|
+
# error first seen at 23:59 that recurred at 00:01 reported zero errors
|
|
137
|
+
# today and two yesterday. Queries::EventVolume counts per-event records
|
|
138
|
+
# instead: occurrence rows, plus storm-shed time buckets, plus the
|
|
139
|
+
# remainder of any group that has neither.
|
|
117
140
|
def event_count_since(since)
|
|
118
|
-
|
|
141
|
+
Queries::EventVolume.in_window(base_scope, since)
|
|
142
|
+
end
|
|
143
|
+
|
|
144
|
+
# The widest-window total, memoised: the stats hash shows it and the
|
|
145
|
+
# completeness predicate needs the same number. One instance answers one
|
|
146
|
+
# call, and this runs on the capture path via the stats broadcast, so a
|
|
147
|
+
# second identical aggregate here is pure waste.
|
|
148
|
+
def month_event_count
|
|
149
|
+
@month_event_count ||= event_count_since(WIDEST_DISPLAYED_WINDOW.ago)
|
|
119
150
|
end
|
|
120
151
|
|
|
121
152
|
def event_count_between(from, to)
|
|
122
|
-
|
|
153
|
+
Queries::EventVolume.in_window(base_scope, from, to)
|
|
123
154
|
end
|
|
124
155
|
|
|
125
156
|
# Occurrence rows carry the user of EACH event. The group's user_id is
|
|
@@ -147,67 +178,171 @@ module RailsErrorDashboard
|
|
|
147
178
|
recorded = occurrence_scope
|
|
148
179
|
.where("#{ErrorOccurrence.table_name}.occurred_at >= ?", Time.current.beginning_of_day)
|
|
149
180
|
.count
|
|
150
|
-
recorded <
|
|
181
|
+
recorded < today_event_count
|
|
151
182
|
rescue StandardError
|
|
152
183
|
false
|
|
153
184
|
end
|
|
154
185
|
|
|
186
|
+
# True when a storm episode overlapping this window degraded without
|
|
187
|
+
# writing its time buckets.
|
|
188
|
+
#
|
|
189
|
+
# Read from the persisted episode rather than recomputed: the flush that
|
|
190
|
+
# knew is long over by the time the dashboard renders, and the flag has
|
|
191
|
+
# to survive a replay. Every guard here is deliberate -- the storm table
|
|
192
|
+
# may not exist, and the column may not be migrated yet on an older host.
|
|
193
|
+
# True when any recorded timing gap overlaps a window this page shows.
|
|
194
|
+
#
|
|
195
|
+
# The page displays today, 7-day AND 30-day figures, so the predicate has
|
|
196
|
+
# to match the WIDEST of them. An earlier version asked only about today
|
|
197
|
+
# and dropped the warning while the weekly and monthly numbers on the
|
|
198
|
+
# same screen still contained the affected events.
|
|
199
|
+
#
|
|
200
|
+
# Read from EventTimingGap rather than the storm episode: the episode is
|
|
201
|
+
# optional (the gate can shed with its breaker closed), and it is written
|
|
202
|
+
# after the counts transaction commits. See the migration for the full
|
|
203
|
+
# history of why this moved.
|
|
204
|
+
def event_timing_incomplete?
|
|
205
|
+
# Nothing left to qualify, so no warning.
|
|
206
|
+
#
|
|
207
|
+
# Gaps are deliberately retained past their own retention while the
|
|
208
|
+
# figures they describe are still shown (see the cleanup job), so a gap
|
|
209
|
+
# can outlive every event it covered -- a group expiring is what
|
|
210
|
+
# removes those events. Warning about a window whose subject matter is
|
|
211
|
+
# gone is noise, and noise trains people to ignore the banner.
|
|
212
|
+
#
|
|
213
|
+
# Asked of surviving GROUPS, not of the event aggregate. The aggregate
|
|
214
|
+
# is the wrong witness here: losing the time buckets is exactly what
|
|
215
|
+
# makes events invisible to it, so an old group that recurs today
|
|
216
|
+
# reports zero for every window (EventVolume can only place an
|
|
217
|
+
# untracked remainder at the group's own occurred_at, which precedes
|
|
218
|
+
# the window -- see untracked_groups) while the events are real, the
|
|
219
|
+
# group is alive and a current gap records them. Reading that zero as
|
|
220
|
+
# "nothing happened" suppressed the banner in precisely the state it
|
|
221
|
+
# exists to announce.
|
|
222
|
+
#
|
|
223
|
+
# last_seen_at is the right evidence because it is what retention
|
|
224
|
+
# deletes on: a group is expired only once it has not been seen for
|
|
225
|
+
# retention_days, so "no group seen in this window" is the same fact as
|
|
226
|
+
# "the covered events are gone" -- and it is indexed, and immune to the
|
|
227
|
+
# timing loss itself.
|
|
228
|
+
#
|
|
229
|
+
# Suppressed only when BOTH witnesses are silent, because neither alone
|
|
230
|
+
# is sufficient and they fail in opposite directions. The aggregate
|
|
231
|
+
# misses an old group's untracked recurrence (the defect above); the
|
|
232
|
+
# liveness check misses the converse, since EventVolume windows
|
|
233
|
+
# occurrence rows and buckets on THEIR OWN timestamps against an
|
|
234
|
+
# unwindowed group set (see group_ids), so a group whose last_seen_at
|
|
235
|
+
# has fallen behind its own event rows can still put events on the page.
|
|
236
|
+
# Requiring both to be empty means the banner can never be dropped
|
|
237
|
+
# while any displayed figure is non-zero, which is the F22 invariant.
|
|
238
|
+
return false unless groups_seen_in_widest_window? || month_event_count.positive?
|
|
239
|
+
|
|
240
|
+
EventTimingGap.affecting?(WIDEST_DISPLAYED_WINDOW.ago, application_id: @application_id)
|
|
241
|
+
rescue StandardError
|
|
242
|
+
false
|
|
243
|
+
end
|
|
244
|
+
|
|
245
|
+
# Whether any error group was last seen inside the widest displayed
|
|
246
|
+
# window. Existence, not a count: one indexed row settles it, and this
|
|
247
|
+
# runs on the capture path via the stats broadcast.
|
|
248
|
+
#
|
|
249
|
+
# The NULL arm covers rows written before last_seen_at existed, whose
|
|
250
|
+
# occurred_at is the only timestamp they have -- the same COALESCE
|
|
251
|
+
# equivalence the retention job documents, written so each arm keeps its
|
|
252
|
+
# own index.
|
|
253
|
+
def groups_seen_in_widest_window?
|
|
254
|
+
cutoff = WIDEST_DISPLAYED_WINDOW.ago
|
|
255
|
+
base_scope.where(
|
|
256
|
+
"last_seen_at >= :cutoff OR (last_seen_at IS NULL AND occurred_at >= :cutoff)", cutoff: cutoff
|
|
257
|
+
).exists?
|
|
258
|
+
end
|
|
259
|
+
|
|
155
260
|
def reopened_count
|
|
156
261
|
return 0 unless ErrorLog.column_names.include?("reopened_at")
|
|
157
262
|
|
|
158
263
|
base_scope.where.not(reopened_at: nil).count
|
|
159
264
|
end
|
|
160
265
|
|
|
266
|
+
# Top error types in the last 7 days, by EVENTS.
|
|
267
|
+
#
|
|
268
|
+
# Filtering groups by first-seen and summing their LIFETIME
|
|
269
|
+
# occurrence_count answered a different question: an August group that
|
|
270
|
+
# recurred today was excluded entirely, so reopening it produced an empty
|
|
271
|
+
# list while the headline total (already on EventVolume) counted the
|
|
272
|
+
# event. Routed through the same primitive as the total so the page's
|
|
273
|
+
# parts and its whole cannot drift apart.
|
|
161
274
|
def top_errors
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
275
|
+
weekly_volume.by_group_attribute(:error_type)
|
|
276
|
+
.sort_by { |_, count| -count }
|
|
277
|
+
.first(10)
|
|
278
|
+
.to_h
|
|
279
|
+
end
|
|
280
|
+
|
|
281
|
+
# One EventVolume for the 7-day window, shared by every weekly breakdown.
|
|
282
|
+
def weekly_volume
|
|
283
|
+
@weekly_volume ||= Queries::EventVolume.new(base_scope, 7.days.ago)
|
|
168
284
|
end
|
|
169
285
|
|
|
170
286
|
# Get 7-day error trend (daily counts)
|
|
287
|
+
# Events per day, placed on the day they HAPPENED. Grouping the ErrorLog
|
|
288
|
+
# table by its own occurred_at put a group's whole lifetime volume on the
|
|
289
|
+
# day it was first seen, so a recurrence never moved the trend.
|
|
290
|
+
#
|
|
291
|
+
# The zero-filled date range is preserved: the chart needs a point for
|
|
292
|
+
# every day in the window, not only the days that had errors.
|
|
171
293
|
def errors_trend_7d
|
|
172
|
-
|
|
173
|
-
.group_by_day(:occurred_at, range: 7.days.ago.to_date..Date.current, default_value: 0)
|
|
174
|
-
.sum(:occurrence_count)
|
|
294
|
+
week_event_counts
|
|
175
295
|
end
|
|
176
296
|
|
|
177
297
|
# Get error counts by severity for last 7 days
|
|
178
298
|
# OPTIMIZED: Use database filtering instead of loading all records into Ruby
|
|
299
|
+
# Weekly events by severity.
|
|
300
|
+
#
|
|
301
|
+
# Folded from the SAME per-type breakdown top_errors uses, rather than
|
|
302
|
+
# four independently scoped sums. Two reasons: it was summing LIFETIME
|
|
303
|
+
# counts of groups born in the window (so a reopened August group scored
|
|
304
|
+
# zero across every severity), and deriving both figures from one
|
|
305
|
+
# breakdown makes them agree by construction instead of by luck --
|
|
306
|
+
# asserted by spec/queries/dashboard_breakdowns_event_volume_spec.rb.
|
|
179
307
|
def errors_by_severity_7d
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
error_type
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
308
|
+
critical = Services::SeverityClassifier::CRITICAL_ERROR_TYPES
|
|
309
|
+
high = Services::SeverityClassifier::HIGH_SEVERITY_ERROR_TYPES
|
|
310
|
+
medium = Services::SeverityClassifier::MEDIUM_SEVERITY_ERROR_TYPES
|
|
311
|
+
|
|
312
|
+
totals = { critical: 0, high: 0, medium: 0, low: 0 }
|
|
313
|
+
weekly_volume.by_group_attribute(:error_type).each do |error_type, count|
|
|
314
|
+
bucket =
|
|
315
|
+
if critical.include?(error_type) then :critical
|
|
316
|
+
elsif high.include?(error_type) then :high
|
|
317
|
+
elsif medium.include?(error_type) then :medium
|
|
318
|
+
else :low
|
|
319
|
+
end
|
|
320
|
+
totals[bucket] += count
|
|
321
|
+
end
|
|
322
|
+
totals
|
|
192
323
|
end
|
|
193
324
|
|
|
194
325
|
# Detect if there's an error spike
|
|
195
326
|
# Uses baselines if available, falls back to simple 2x average
|
|
327
|
+
#
|
|
328
|
+
# Memoised: the stats hash asks twice (spike_detected and spike_info), and
|
|
329
|
+
# this runs from the live stats broadcast inside the capture path.
|
|
196
330
|
def spike_detected?
|
|
197
|
-
return
|
|
331
|
+
return @spike_detected if defined?(@spike_detected)
|
|
198
332
|
|
|
199
|
-
|
|
333
|
+
@spike_detected = compute_spike_detected
|
|
334
|
+
end
|
|
200
335
|
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
336
|
+
def compute_spike_detected
|
|
337
|
+
trend = errors_trend_7d
|
|
338
|
+
return false if trend.empty?
|
|
339
|
+
return true if baseline_anomalies.any?
|
|
205
340
|
|
|
206
341
|
# Fall back to simple 2x average detection
|
|
207
|
-
avg_count =
|
|
342
|
+
avg_count = trend.values.sum / 7.0
|
|
208
343
|
return false if avg_count.zero?
|
|
209
344
|
|
|
210
|
-
|
|
345
|
+
today_event_count >= (avg_count * 2)
|
|
211
346
|
end
|
|
212
347
|
|
|
213
348
|
# Get spike information
|
|
@@ -215,7 +350,7 @@ module RailsErrorDashboard
|
|
|
215
350
|
def spike_info
|
|
216
351
|
return nil unless spike_detected?
|
|
217
352
|
|
|
218
|
-
today_count =
|
|
353
|
+
today_count = today_event_count
|
|
219
354
|
avg_count = (errors_trend_7d.values.sum / 7.0).round(1)
|
|
220
355
|
|
|
221
356
|
info = {
|
|
@@ -226,46 +361,46 @@ module RailsErrorDashboard
|
|
|
226
361
|
}
|
|
227
362
|
|
|
228
363
|
# Add baseline info if available
|
|
229
|
-
baseline_info = baseline_anomaly_info
|
|
364
|
+
baseline_info = baseline_anomaly_info
|
|
230
365
|
info.merge!(baseline_info) if baseline_info.present?
|
|
231
366
|
|
|
232
367
|
info
|
|
233
368
|
end
|
|
234
369
|
|
|
235
|
-
#
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
Queries::
|
|
242
|
-
|
|
370
|
+
# Today, yesterday and the 7-day total all come from ONE daily breakdown.
|
|
371
|
+
# Each window is three queries (occurrence rows, shed buckets, untracked
|
|
372
|
+
# remainder), and this method runs on the capture path via the stats
|
|
373
|
+
# broadcast, so asking per window multiplied the query count.
|
|
374
|
+
def week_event_counts
|
|
375
|
+
@week_event_counts ||= begin
|
|
376
|
+
counts = Queries::EventVolume.by_day(base_scope, 7.days.ago)
|
|
377
|
+
(7.days.ago.to_date..Date.current).to_h { |day| [ day, counts[day] || 0 ] }
|
|
243
378
|
end
|
|
244
379
|
end
|
|
245
380
|
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
381
|
+
def today_event_count
|
|
382
|
+
@today_event_count ||= week_event_counts[Date.current].to_i
|
|
383
|
+
end
|
|
249
384
|
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
385
|
+
# Every anomalous (error_type, platform) pair, from a fixed number of
|
|
386
|
+
# queries (BaselineStats.current_anomalies), loaded once per call. This
|
|
387
|
+
# used to be about six queries per distinct pair, run twice.
|
|
388
|
+
def baseline_anomalies
|
|
389
|
+
return @baseline_anomalies if defined?(@baseline_anomalies)
|
|
255
390
|
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
}
|
|
263
|
-
end.compact
|
|
391
|
+
@baseline_anomalies = if defined?(Queries::BaselineStats)
|
|
392
|
+
Queries::BaselineStats.current_anomalies(sensitivity: 2, application_id: @application_id)
|
|
393
|
+
else
|
|
394
|
+
[]
|
|
395
|
+
end
|
|
396
|
+
end
|
|
264
397
|
|
|
265
|
-
|
|
398
|
+
# Get baseline anomaly information
|
|
399
|
+
def baseline_anomaly_info
|
|
400
|
+
return nil if baseline_anomalies.empty?
|
|
266
401
|
|
|
267
402
|
# Return info about worst anomaly
|
|
268
|
-
worst =
|
|
403
|
+
worst = baseline_anomalies.max_by { |a| a[:std_devs_above] || 0 }
|
|
269
404
|
{
|
|
270
405
|
baseline_detected: true,
|
|
271
406
|
anomaly_error_type: worst[:error_type],
|
|
@@ -283,7 +418,7 @@ module RailsErrorDashboard
|
|
|
283
418
|
# make a real failure percentage from, so the honest figure is the rate
|
|
284
419
|
# itself, uncapped, labelled with its unit.
|
|
285
420
|
def error_rate
|
|
286
|
-
today_events =
|
|
421
|
+
today_events = today_event_count
|
|
287
422
|
return 0.0 if today_events.zero?
|
|
288
423
|
|
|
289
424
|
hours_today = ((Time.current - Time.current.beginning_of_day) / 1.hour).round(1)
|
|
@@ -299,11 +434,12 @@ module RailsErrorDashboard
|
|
|
299
434
|
# affected user. Storm count-only events create no occurrence row, so
|
|
300
435
|
# this is a floor during a storm -- affected_users_incomplete? says when.
|
|
301
436
|
def affected_users_today
|
|
302
|
-
distinct_affected_users(Time.current.beginning_of_day, nil)
|
|
437
|
+
@affected_users_today ||= distinct_affected_users(Time.current.beginning_of_day, nil)
|
|
303
438
|
end
|
|
304
439
|
|
|
305
440
|
def affected_users_yesterday
|
|
306
|
-
|
|
441
|
+
@affected_users_yesterday ||=
|
|
442
|
+
distinct_affected_users(1.day.ago.beginning_of_day, Time.current.beginning_of_day)
|
|
307
443
|
end
|
|
308
444
|
|
|
309
445
|
def distinct_affected_users(from, to)
|
|
@@ -334,8 +470,14 @@ module RailsErrorDashboard
|
|
|
334
470
|
|
|
335
471
|
# Calculate percentage change in errors (today vs yesterday)
|
|
336
472
|
def trend_percentage
|
|
337
|
-
|
|
338
|
-
|
|
473
|
+
return @trend_percentage if defined?(@trend_percentage)
|
|
474
|
+
|
|
475
|
+
@trend_percentage = compute_trend_percentage
|
|
476
|
+
end
|
|
477
|
+
|
|
478
|
+
def compute_trend_percentage
|
|
479
|
+
today = today_event_count
|
|
480
|
+
yesterday = week_event_counts[Date.current - 1].to_i
|
|
339
481
|
|
|
340
482
|
return 0.0 if today.zero? && yesterday.zero?
|
|
341
483
|
return 100.0 if yesterday.zero? && today.positive?
|
|
@@ -224,12 +224,15 @@ module RailsErrorDashboard
|
|
|
224
224
|
previous_start = @start_date
|
|
225
225
|
previous_end = current_start
|
|
226
226
|
|
|
227
|
-
|
|
227
|
+
# Both periods come from base_query, which carries the application
|
|
228
|
+
# filter (and the window start). Querying ErrorLog directly made this
|
|
229
|
+
# the one panel on an application-filtered page that counted every app.
|
|
230
|
+
current_errors = base_query
|
|
228
231
|
.where("occurred_at >= ?", current_start)
|
|
229
232
|
.count
|
|
230
233
|
|
|
231
|
-
previous_errors =
|
|
232
|
-
.where("occurred_at
|
|
234
|
+
previous_errors = base_query
|
|
235
|
+
.where("occurred_at < ?", previous_end)
|
|
233
236
|
.count
|
|
234
237
|
|
|
235
238
|
change_percentage = if previous_errors > 0
|
|
@@ -5,6 +5,9 @@ module RailsErrorDashboard
|
|
|
5
5
|
# Query: Fetch errors with filtering and pagination
|
|
6
6
|
# This is a read operation that returns a filtered collection of errors
|
|
7
7
|
class ErrorsList
|
|
8
|
+
# Escape character for LIKE patterns; see filter_by_search.
|
|
9
|
+
LIKE_ESCAPE = "!"
|
|
10
|
+
|
|
8
11
|
def self.call(filters = {})
|
|
9
12
|
new(filters).call
|
|
10
13
|
end
|
|
@@ -129,9 +132,19 @@ module RailsErrorDashboard
|
|
|
129
132
|
else
|
|
130
133
|
# Fall back to LIKE for SQLite/MySQL - search across all relevant fields
|
|
131
134
|
# Use LOWER() for case-insensitive search
|
|
132
|
-
|
|
135
|
+
#
|
|
136
|
+
# % and _ in the term are escaped so they match themselves: a search
|
|
137
|
+
# for "snake_case" must not match "snakeXcase", and "%" must not
|
|
138
|
+
# match every row. The escape character is named explicitly because
|
|
139
|
+
# SQLite has no default one, and it is "!" rather than a backslash
|
|
140
|
+
# because a backslash inside a quoted SQL literal means different
|
|
141
|
+
# things to MySQL and to everyone else.
|
|
142
|
+
escaped = ActiveRecord::Base.sanitize_sql_like(@filters[:search].to_s, LIKE_ESCAPE)
|
|
143
|
+
search_pattern = "%#{escaped}%"
|
|
133
144
|
query.where(
|
|
134
|
-
"LOWER(message) LIKE LOWER(?)
|
|
145
|
+
"LOWER(message) LIKE LOWER(?) ESCAPE '#{LIKE_ESCAPE}' " \
|
|
146
|
+
"OR LOWER(COALESCE(backtrace, '')) LIKE LOWER(?) ESCAPE '#{LIKE_ESCAPE}' " \
|
|
147
|
+
"OR LOWER(error_type) LIKE LOWER(?) ESCAPE '#{LIKE_ESCAPE}'",
|
|
135
148
|
search_pattern, search_pattern, search_pattern
|
|
136
149
|
)
|
|
137
150
|
end
|