rails_error_dashboard 0.12.0 → 0.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/app/controllers/rails_error_dashboard/application_controller.rb +92 -11
- data/app/controllers/rails_error_dashboard/errors_controller.rb +111 -34
- data/app/controllers/rails_error_dashboard/webhooks_controller.rb +3 -0
- data/app/helpers/rails_error_dashboard/application_helper.rb +46 -1
- data/app/helpers/rails_error_dashboard/backtrace_helper.rb +8 -0
- data/app/jobs/rails_error_dashboard/concerns/plain_channel_message.rb +71 -0
- data/app/jobs/rails_error_dashboard/notification_burst_summary_job.rb +55 -0
- data/app/jobs/rails_error_dashboard/retention_cleanup_job.rb +68 -1
- data/app/jobs/rails_error_dashboard/storm_notification_job.rb +8 -44
- data/app/models/rails_error_dashboard/error_baseline.rb +12 -9
- data/app/models/rails_error_dashboard/error_comment.rb +0 -5
- data/app/models/rails_error_dashboard/error_log.rb +9 -3
- data/app/models/rails_error_dashboard/error_logs_record.rb +34 -0
- data/app/models/rails_error_dashboard/error_occurrence.rb +9 -1
- data/app/views/layouts/rails_error_dashboard.html.erb +11 -3
- data/app/views/rails_error_dashboard/errors/_discussion.html.erb +18 -11
- data/app/views/rails_error_dashboard/errors/_issue_section.html.erb +22 -8
- data/app/views/rails_error_dashboard/errors/_sidebar_metadata.html.erb +15 -6
- data/app/views/rails_error_dashboard/errors/analytics.html.erb +14 -8
- data/app/views/rails_error_dashboard/errors/diagnostic_dumps.html.erb +1 -1
- data/app/views/rails_error_dashboard/errors/index.html.erb +6 -1
- data/app/views/rails_error_dashboard/errors/platform_comparison.html.erb +7 -7
- data/app/views/rails_error_dashboard/errors/releases.html.erb +2 -2
- data/app/views/rails_error_dashboard/errors/settings.html.erb +2 -0
- data/config/locales/de.yml +28 -0
- data/config/locales/en.yml +35 -0
- data/config/locales/es.yml +28 -0
- data/config/locales/fr.yml +28 -0
- data/config/locales/it.yml +28 -0
- data/config/locales/ja.yml +28 -0
- data/config/locales/pl.yml +28 -0
- data/config/locales/pt-BR.yml +28 -0
- data/config/locales/ru.yml +28 -0
- data/config/locales/uk.yml +28 -0
- data/config/locales/zh-CN.yml +28 -0
- data/db/migrate/20260917000001_add_last_notified_at_to_error_logs.rb +40 -0
- data/lib/generators/rails_error_dashboard/install/templates/initializer.rb +12 -1
- data/lib/rails_error_dashboard/commands/assign_error.rb +12 -2
- data/lib/rails_error_dashboard/commands/backfill_environments.rb +2 -0
- data/lib/rails_error_dashboard/commands/backfill_resolved_at.rb +42 -0
- data/lib/rails_error_dashboard/commands/batch_delete_errors.rb +1 -0
- data/lib/rails_error_dashboard/commands/batch_mute_errors.rb +2 -0
- data/lib/rails_error_dashboard/commands/batch_resolve_errors.rb +2 -0
- data/lib/rails_error_dashboard/commands/batch_unmute_errors.rb +2 -0
- data/lib/rails_error_dashboard/commands/find_or_increment_error.rb +39 -7
- data/lib/rails_error_dashboard/commands/flush_rack_attack_events.rb +3 -1
- data/lib/rails_error_dashboard/commands/flush_storm_counts.rb +27 -4
- data/lib/rails_error_dashboard/commands/flush_swallowed_exceptions.rb +5 -2
- data/lib/rails_error_dashboard/commands/link_existing_issue.rb +1 -0
- data/lib/rails_error_dashboard/commands/log_error.rb +82 -23
- data/lib/rails_error_dashboard/commands/mute_error.rb +1 -0
- data/lib/rails_error_dashboard/commands/resolve_error.rb +2 -0
- data/lib/rails_error_dashboard/commands/scrub_invalid_encoding.rb +104 -0
- data/lib/rails_error_dashboard/commands/snooze_error.rb +34 -10
- data/lib/rails_error_dashboard/commands/unmute_error.rb +1 -0
- data/lib/rails_error_dashboard/commands/update_error_priority.rb +23 -2
- data/lib/rails_error_dashboard/commands/update_error_status.rb +33 -6
- data/lib/rails_error_dashboard/configuration.rb +19 -1
- data/lib/rails_error_dashboard/engine.rb +15 -0
- data/lib/rails_error_dashboard/queries/analytics_stats.rb +107 -26
- data/lib/rails_error_dashboard/queries/baseline_stats.rb +107 -0
- data/lib/rails_error_dashboard/queries/dashboard_stats.rb +55 -50
- data/lib/rails_error_dashboard/queries/error_correlation.rb +6 -3
- data/lib/rails_error_dashboard/queries/errors_list.rb +15 -2
- data/lib/rails_error_dashboard/queries/similar_errors.rb +1 -1
- data/lib/rails_error_dashboard/services/analytics_cache_manager.rb +43 -18
- data/lib/rails_error_dashboard/services/backtrace_processor.rb +3 -1
- data/lib/rails_error_dashboard/services/cause_chain_extractor.rb +3 -1
- data/lib/rails_error_dashboard/services/codeberg_issue_client.rb +13 -2
- data/lib/rails_error_dashboard/services/diagnostic_dump_generator.rb +5 -3
- data/lib/rails_error_dashboard/services/encoding_sanitizer.rb +80 -0
- data/lib/rails_error_dashboard/services/error_broadcaster.rb +180 -32
- data/lib/rails_error_dashboard/services/error_hash_generator.rb +9 -3
- data/lib/rails_error_dashboard/services/error_notification_dispatcher.rb +13 -0
- data/lib/rails_error_dashboard/services/exception_filter.rb +55 -0
- data/lib/rails_error_dashboard/services/git_head_reader.rb +102 -0
- data/lib/rails_error_dashboard/services/notification_throttler.rb +172 -28
- data/lib/rails_error_dashboard/services/sensitive_data_filter.rb +47 -1
- data/lib/rails_error_dashboard/services/storm_protection/circuit_breaker.rb +59 -3
- data/lib/rails_error_dashboard/services/storm_protection/fingerprint_buckets.rb +25 -2
- data/lib/rails_error_dashboard/services/storm_protection/gate.rb +32 -5
- data/lib/rails_error_dashboard/services/swallowed_exception_tracker.rb +99 -23
- data/lib/rails_error_dashboard/services/url_safety.rb +40 -0
- data/lib/rails_error_dashboard/subscribers/issue_tracker_subscriber.rb +11 -2
- data/lib/rails_error_dashboard/value_objects/error_context.rb +3 -1
- data/lib/rails_error_dashboard/version.rb +1 -1
- data/lib/rails_error_dashboard.rb +31 -0
- data/lib/tasks/error_dashboard.rake +54 -4
- metadata +10 -2
|
@@ -157,6 +157,8 @@ module RailsErrorDashboard
|
|
|
157
157
|
attr_accessor :notification_minimum_severity # Minimum severity to notify (default: :low = notify all)
|
|
158
158
|
attr_accessor :notification_cooldown_minutes # Per-error cooldown in minutes (default: 5, 0 = disabled)
|
|
159
159
|
attr_accessor :notification_threshold_alerts # Occurrence milestones that trigger notification (default: [10, 50, 100, 500, 1000])
|
|
160
|
+
attr_accessor :notification_burst_limit # Max FIRST-OCCURRENCE notifications per window, per process (default: 10, 0 = no cap)
|
|
161
|
+
attr_accessor :notification_burst_window_seconds # Length of that window in seconds (default: 60)
|
|
160
162
|
|
|
161
163
|
# Breadcrumbs (request activity trail)
|
|
162
164
|
attr_accessor :enable_breadcrumbs # Master switch (default: false)
|
|
@@ -302,7 +304,8 @@ module RailsErrorDashboard
|
|
|
302
304
|
|
|
303
305
|
@use_separate_database = ENV.fetch("USE_SEPARATE_ERROR_DB", "false") == "true"
|
|
304
306
|
|
|
305
|
-
# Retention policy - days
|
|
307
|
+
# Retention policy - days an error may go unseen (last_seen_at) before it is
|
|
308
|
+
# deleted automatically (default: 90). An error still occurring is kept.
|
|
306
309
|
# Set to nil to keep errors forever (not recommended for production)
|
|
307
310
|
# Schedule cleanup: RailsErrorDashboard::RetentionCleanupJob.perform_later
|
|
308
311
|
@retention_days = 90
|
|
@@ -384,6 +387,8 @@ module RailsErrorDashboard
|
|
|
384
387
|
@notification_minimum_severity = :low # Notify on all severities (current behavior)
|
|
385
388
|
@notification_cooldown_minutes = 5 # 5 min cooldown per error_hash (0 = disabled)
|
|
386
389
|
@notification_threshold_alerts = [ 10, 50, 100, 500, 1000 ] # Occurrence milestones
|
|
390
|
+
@notification_burst_limit = 10 # New-error notifications per window, per process (0 = no cap)
|
|
391
|
+
@notification_burst_window_seconds = 60 # One summary message replaces the rest of the window
|
|
387
392
|
|
|
388
393
|
# Breadcrumbs defaults - OFF by default (opt-in)
|
|
389
394
|
@enable_breadcrumbs = false # Master switch
|
|
@@ -854,6 +859,19 @@ module RailsErrorDashboard
|
|
|
854
859
|
errors << "notification_threshold_alerts must be an Array (got: #{notification_threshold_alerts.class})"
|
|
855
860
|
end
|
|
856
861
|
|
|
862
|
+
# Validate the first-occurrence burst cap (non-negative integers; 0 or nil
|
|
863
|
+
# turns the cap off)
|
|
864
|
+
{
|
|
865
|
+
notification_burst_limit: notification_burst_limit,
|
|
866
|
+
notification_burst_window_seconds: notification_burst_window_seconds
|
|
867
|
+
}.each do |name, value|
|
|
868
|
+
next if value.nil?
|
|
869
|
+
|
|
870
|
+
unless value.is_a?(Integer) && value >= 0
|
|
871
|
+
errors << "#{name} must be a non-negative Integer (got: #{value.inspect})"
|
|
872
|
+
end
|
|
873
|
+
end
|
|
874
|
+
|
|
857
875
|
# Log warnings (non-fatal issues)
|
|
858
876
|
warnings.each do |warning|
|
|
859
877
|
Rails.logger.warn "[Rails Error Dashboard] #{warning}" if defined?(Rails) && Rails.respond_to?(:logger) && Rails.logger
|
|
@@ -71,6 +71,11 @@ module RailsErrorDashboard
|
|
|
71
71
|
next
|
|
72
72
|
end
|
|
73
73
|
|
|
74
|
+
# Resolve the running commit now (three small file reads, memoised, never
|
|
75
|
+
# raises) so that no capture ever pays for it. Skipped when the SHA is
|
|
76
|
+
# configured: then it is never consulted.
|
|
77
|
+
RailsErrorDashboard.detected_git_sha if RailsErrorDashboard.configuration.git_sha.blank?
|
|
78
|
+
|
|
74
79
|
if RailsErrorDashboard.configuration.enable_error_subscriber
|
|
75
80
|
Rails.error.subscribe(RailsErrorDashboard::ErrorReporter.new)
|
|
76
81
|
end
|
|
@@ -188,6 +193,16 @@ module RailsErrorDashboard
|
|
|
188
193
|
# Enable TracePoint(:raise) + TracePoint(:rescue) for swallowed exception detection
|
|
189
194
|
if RailsErrorDashboard.configuration.detect_swallowed_exceptions
|
|
190
195
|
RailsErrorDashboard::Services::SwallowedExceptionTracker.enable!
|
|
196
|
+
|
|
197
|
+
# Drain buffered counts at the end of every request and job. Without
|
|
198
|
+
# this the buffer is only ever drained by a LATER rescue on the SAME
|
|
199
|
+
# thread, so a swallowed exception that happens once stays invisible
|
|
200
|
+
# until the process exits. to_complete fires after the response body
|
|
201
|
+
# is closed, so it never delays a request (safety rule 2); the flush is
|
|
202
|
+
# deadline-gated, so a flood is still one write per interval.
|
|
203
|
+
Rails.application.executor.to_complete do
|
|
204
|
+
RailsErrorDashboard::Services::SwallowedExceptionTracker.flush_if_due!
|
|
205
|
+
end
|
|
191
206
|
end
|
|
192
207
|
|
|
193
208
|
# Import crash files from previous process death, then register at_exit hook
|
|
@@ -17,7 +17,7 @@ module RailsErrorDashboard
|
|
|
17
17
|
|
|
18
18
|
def call
|
|
19
19
|
# Cache analytics data for 5 minutes to reduce database load
|
|
20
|
-
# Cache key includes days parameter and
|
|
20
|
+
# Cache key includes the days parameter and the cache generation
|
|
21
21
|
Rails.cache.fetch(cache_key, expires_in: 5.minutes) do
|
|
22
22
|
{
|
|
23
23
|
days: @days,
|
|
@@ -41,13 +41,14 @@ module RailsErrorDashboard
|
|
|
41
41
|
# - Query class name
|
|
42
42
|
# - Days parameter (different time ranges = different caches)
|
|
43
43
|
# - Application ID (per-app caching)
|
|
44
|
-
# -
|
|
44
|
+
# - The cache generation (bumped by user actions; see AnalyticsCacheManager).
|
|
45
|
+
# Captures do not bump it: they rely on the 5-minute TTL.
|
|
45
46
|
# - Start date (ensures correct time window)
|
|
46
47
|
[
|
|
47
48
|
"analytics_stats",
|
|
48
49
|
@days,
|
|
49
50
|
@application_id || "all",
|
|
50
|
-
|
|
51
|
+
Services::AnalyticsCacheManager.generation,
|
|
51
52
|
@start_date.to_date.to_s
|
|
52
53
|
].join("/")
|
|
53
54
|
end
|
|
@@ -64,29 +65,45 @@ module RailsErrorDashboard
|
|
|
64
65
|
base_scope.where("occurred_at >= ?", @start_date)
|
|
65
66
|
end
|
|
66
67
|
|
|
68
|
+
# Two counting units, kept distinct on purpose.
|
|
69
|
+
#
|
|
70
|
+
# An ErrorLog row is a GROUP; its occurrence_count says how many times
|
|
71
|
+
# that error actually happened. Volume figures are EVENTS (the sum), so
|
|
72
|
+
# this page agrees with the Overview, which counts the same way. Resolved
|
|
73
|
+
# and unresolved stay GROUP counts -- a group is the thing that gets
|
|
74
|
+
# resolved, and an event cannot be.
|
|
67
75
|
def error_statistics
|
|
68
76
|
{
|
|
69
|
-
total:
|
|
77
|
+
total: event_count,
|
|
78
|
+
total_groups: base_query.count,
|
|
70
79
|
unresolved: base_query.unresolved.count,
|
|
71
|
-
|
|
72
|
-
|
|
80
|
+
resolved: base_query.resolved.count,
|
|
81
|
+
by_type: base_query.group(:error_type).sum(:occurrence_count).sort_by { |_, count| -count }.to_h,
|
|
82
|
+
by_day: base_query.group("DATE(occurred_at)").sum(:occurrence_count),
|
|
83
|
+
affected_users_incomplete: affected_users_incomplete?
|
|
73
84
|
}
|
|
74
85
|
end
|
|
75
86
|
|
|
87
|
+
# Total EVENTS in the window. dashboard_stats.rb and
|
|
88
|
+
# platform_comparison.rb:165 count the same way.
|
|
89
|
+
def event_count
|
|
90
|
+
base_query.sum(:occurrence_count)
|
|
91
|
+
end
|
|
92
|
+
|
|
76
93
|
def errors_over_time
|
|
77
|
-
base_query.group_by_day(:occurred_at).
|
|
94
|
+
base_query.group_by_day(:occurred_at).sum(:occurrence_count)
|
|
78
95
|
end
|
|
79
96
|
|
|
80
97
|
def errors_by_type
|
|
81
98
|
base_query.group(:error_type)
|
|
82
|
-
.
|
|
99
|
+
.sum(:occurrence_count)
|
|
83
100
|
.sort_by { |_, count| -count }
|
|
84
101
|
.first(10)
|
|
85
102
|
.to_h
|
|
86
103
|
end
|
|
87
104
|
|
|
88
105
|
def errors_by_platform
|
|
89
|
-
base_query.group(:platform).
|
|
106
|
+
base_query.group(:platform).sum(:occurrence_count)
|
|
90
107
|
end
|
|
91
108
|
|
|
92
109
|
# NULL (captured before the column existed) is reported under :unknown
|
|
@@ -94,25 +111,81 @@ module RailsErrorDashboard
|
|
|
94
111
|
def errors_by_environment
|
|
95
112
|
return {} unless ErrorLog.column_names.include?("environment")
|
|
96
113
|
|
|
97
|
-
base_query.group(:environment).
|
|
114
|
+
base_query.group(:environment).sum(:occurrence_count).transform_keys { |env| env.nil? ? :unknown : env }
|
|
98
115
|
end
|
|
99
116
|
|
|
100
117
|
def errors_by_hour
|
|
101
118
|
# group_by_hour_of_day buckets into 0..23 to show diurnal patterns
|
|
102
119
|
# (when in the day errors peak). The chart title says "Errors by Hour
|
|
103
120
|
# of Day" — group_by_hour produced a chronological time series instead.
|
|
104
|
-
base_query.group_by_hour_of_day(:occurred_at).
|
|
121
|
+
base_query.group_by_hour_of_day(:occurred_at).sum(:occurrence_count)
|
|
105
122
|
end
|
|
106
123
|
|
|
124
|
+
# Events per user, counted from OCCURRENCE rows.
|
|
125
|
+
#
|
|
126
|
+
# The group's user_id is overwritten by each new occurrence, so grouping
|
|
127
|
+
# ErrorLog by it attributed a whole group to whoever happened to hit it
|
|
128
|
+
# last -- at most one user per group, and their "count" was a number of
|
|
129
|
+
# groups. Occurrence rows carry the user of each individual event.
|
|
130
|
+
#
|
|
131
|
+
# Storm count-only events create no occurrence row, so during a storm
|
|
132
|
+
# this is a floor; affected_users_incomplete? says when.
|
|
107
133
|
def top_affected_users
|
|
108
134
|
user_model = RailsErrorDashboard.configuration.user_model
|
|
109
135
|
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
136
|
+
counts = user_event_counts
|
|
137
|
+
counts.sort_by { |_, count| -count }
|
|
138
|
+
.first(10)
|
|
139
|
+
.map { |user_id, count| { user_id: user_id, email: find_user_email(user_id, user_model), count: count } }
|
|
140
|
+
end
|
|
141
|
+
|
|
142
|
+
def user_event_counts
|
|
143
|
+
return group_user_counts unless occurrences_available?
|
|
144
|
+
|
|
145
|
+
occurrences = ErrorOccurrence.table_name
|
|
146
|
+
counts = occurrence_scope.where("#{occurrences}.occurred_at >= ?", @start_date)
|
|
147
|
+
.where.not(occurrences => { user_id: nil })
|
|
148
|
+
.group("#{occurrences}.user_id")
|
|
149
|
+
.count
|
|
150
|
+
|
|
151
|
+
# A user whose events predate occurrence tracking -- or whose events
|
|
152
|
+
# were shed by storm protection, which writes no occurrence row -- still
|
|
153
|
+
# belongs in the table. Fall back to the group's own user for those,
|
|
154
|
+
# taking whichever count is larger. UserImpactSummary merges the same way.
|
|
155
|
+
group_user_counts.merge(counts) { |_user, group_count, occurrence_count| [ group_count, occurrence_count ].max }
|
|
156
|
+
rescue StandardError
|
|
157
|
+
group_user_counts
|
|
158
|
+
end
|
|
159
|
+
|
|
160
|
+
def group_user_counts
|
|
161
|
+
base_query.where.not(user_id: nil).group(:user_id).sum(:occurrence_count)
|
|
162
|
+
end
|
|
163
|
+
|
|
164
|
+
# Occurrence rows joined back to error_logs, so the application filter
|
|
165
|
+
# (which the occurrence table has no column for) still applies.
|
|
166
|
+
def occurrence_scope
|
|
167
|
+
scope = ErrorOccurrence.joins(:error_log)
|
|
168
|
+
scope = scope.where(ErrorLog.table_name => { application_id: @application_id }) if @application_id.present?
|
|
169
|
+
scope
|
|
170
|
+
end
|
|
171
|
+
|
|
172
|
+
def occurrences_available?
|
|
173
|
+
defined?(ErrorOccurrence) && ErrorOccurrence.table_exists?
|
|
174
|
+
rescue StandardError
|
|
175
|
+
false
|
|
176
|
+
end
|
|
177
|
+
|
|
178
|
+
# True when the window holds more events than recorded occurrences --
|
|
179
|
+
# storm shedding dropped per-event rows, so any occurrence-derived
|
|
180
|
+
# figure (the affected-user table) is a floor, not a total.
|
|
181
|
+
def affected_users_incomplete?
|
|
182
|
+
return false unless occurrences_available?
|
|
183
|
+
|
|
184
|
+
occurrences = ErrorOccurrence.table_name
|
|
185
|
+
recorded = occurrence_scope.where("#{occurrences}.occurred_at >= ?", @start_date).count
|
|
186
|
+
recorded < event_count
|
|
187
|
+
rescue StandardError
|
|
188
|
+
false
|
|
116
189
|
end
|
|
117
190
|
|
|
118
191
|
def find_user_email(user_id, user_model)
|
|
@@ -122,23 +195,31 @@ module RailsErrorDashboard
|
|
|
122
195
|
"User ##{user_id}"
|
|
123
196
|
end
|
|
124
197
|
|
|
198
|
+
# Resolution rate is GROUPS over GROUPS.
|
|
199
|
+
#
|
|
200
|
+
# Volume is now measured in events, but resolving is something that
|
|
201
|
+
# happens to a group, so dividing resolved groups by total events would
|
|
202
|
+
# collapse the rate toward zero the moment any error recurred. The
|
|
203
|
+
# Overview computes this the same way (resolved / resolved + unresolved),
|
|
204
|
+
# so both pages report one rate.
|
|
205
|
+
#
|
|
206
|
+
# Same scoped relation on both sides: an unscoped numerator counted every
|
|
207
|
+
# application's resolved errors against one application's total, and the
|
|
208
|
+
# "rate" went past 100%.
|
|
125
209
|
def resolution_rate
|
|
126
|
-
total = error_statistics[:total]
|
|
127
|
-
return 0 if total.zero?
|
|
128
|
-
|
|
129
|
-
# Same scoped relation as the denominator: an unscoped numerator
|
|
130
|
-
# counted every application's resolved errors against one
|
|
131
|
-
# application's total, and the "rate" went past 100%.
|
|
132
210
|
resolved_count = base_query.resolved.count
|
|
133
|
-
|
|
211
|
+
total_groups = resolved_count + base_query.unresolved.count
|
|
212
|
+
return 0 if total_groups.zero?
|
|
213
|
+
|
|
214
|
+
((resolved_count.to_f / total_groups) * 100).round(1)
|
|
134
215
|
end
|
|
135
216
|
|
|
136
217
|
def mobile_errors_count
|
|
137
|
-
base_query.where(platform: [ "iOS", "Android" ]).
|
|
218
|
+
base_query.where(platform: [ "iOS", "Android" ]).sum(:occurrence_count)
|
|
138
219
|
end
|
|
139
220
|
|
|
140
221
|
def api_errors_count
|
|
141
|
-
base_query.where("platform IS NULL OR platform = ?", "API").
|
|
222
|
+
base_query.where("platform IS NULL OR platform = ?", "API").sum(:occurrence_count)
|
|
142
223
|
end
|
|
143
224
|
|
|
144
225
|
# Pattern insights for top error types
|
|
@@ -23,6 +23,113 @@ module RailsErrorDashboard
|
|
|
23
23
|
new(error_type, platform).weekly_baseline
|
|
24
24
|
end
|
|
25
25
|
|
|
26
|
+
# Pairs considered by .current_anomalies, busiest first. A bound, so the
|
|
27
|
+
# check costs the same whether 5 or 50,000 error types are stored.
|
|
28
|
+
MAX_ANOMALY_PAIRS = 500
|
|
29
|
+
BASELINE_PRECEDENCE = %w[hourly daily weekly].freeze
|
|
30
|
+
|
|
31
|
+
# Every (error_type, platform) pair that is anomalous RIGHT NOW, for all
|
|
32
|
+
# pairs at once. Same rule as #check_current_anomaly -- the first baseline
|
|
33
|
+
# available among hourly / daily / weekly, compared against this hour /
|
|
34
|
+
# today / this week -- but in a fixed number of queries instead of about six
|
|
35
|
+
# per pair. DashboardStats calls this from the live stats broadcast, which
|
|
36
|
+
# runs inside the host app's capture path.
|
|
37
|
+
#
|
|
38
|
+
# Only pairs with an event this week are looked at: a pair with none has a
|
|
39
|
+
# count of zero, which no baseline can call anomalous.
|
|
40
|
+
#
|
|
41
|
+
# @return [Array<Hash>] error_type, platform, count, level, std_devs_above,
|
|
42
|
+
# baseline_type. Empty on any failure -- never raises.
|
|
43
|
+
def self.current_anomalies(sensitivity: 2, application_id: nil)
|
|
44
|
+
return [] unless defined?(ErrorBaseline) && ErrorBaseline.table_exists?
|
|
45
|
+
|
|
46
|
+
counts = current_counts_by_pair(application_id: application_id)
|
|
47
|
+
return [] if counts.empty?
|
|
48
|
+
|
|
49
|
+
baselines = latest_baselines_for(counts.keys)
|
|
50
|
+
|
|
51
|
+
counts.filter_map do |(error_type, platform), windows|
|
|
52
|
+
kind = BASELINE_PRECEDENCE.find { |type| baselines[[ error_type, platform, type ]] }
|
|
53
|
+
next unless kind
|
|
54
|
+
|
|
55
|
+
baseline = baselines[[ error_type, platform, kind ]]
|
|
56
|
+
# A flat history has no spread to measure against; dividing by it
|
|
57
|
+
# makes any count above the mean infinitely anomalous.
|
|
58
|
+
next if baseline.std_dev.nil? || baseline.std_dev.zero?
|
|
59
|
+
|
|
60
|
+
count = windows.fetch(kind.to_sym)
|
|
61
|
+
level = baseline.anomaly_level(count, sensitivity: sensitivity)
|
|
62
|
+
next unless level
|
|
63
|
+
|
|
64
|
+
{
|
|
65
|
+
error_type: error_type,
|
|
66
|
+
platform: platform,
|
|
67
|
+
count: count,
|
|
68
|
+
level: level,
|
|
69
|
+
std_devs_above: baseline.std_devs_above_mean(count),
|
|
70
|
+
baseline_type: kind
|
|
71
|
+
}
|
|
72
|
+
end
|
|
73
|
+
rescue => e
|
|
74
|
+
RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] current_anomalies failed: #{e.class}: #{e.message}")
|
|
75
|
+
[]
|
|
76
|
+
end
|
|
77
|
+
|
|
78
|
+
# { [error_type, platform] => { hourly:, daily:, weekly: } } in ONE grouped
|
|
79
|
+
# query, counted in the units the baselines were built from (occurrence
|
|
80
|
+
# rows when that table exists). The week is the widest window, so it is
|
|
81
|
+
# the WHERE; the day and the hour are conditional sums inside it.
|
|
82
|
+
def self.current_counts_by_pair(application_id: nil)
|
|
83
|
+
logs = ErrorLog.table_name
|
|
84
|
+
column = Services::BaselineCalculator.time_column
|
|
85
|
+
now = Time.current
|
|
86
|
+
|
|
87
|
+
relation = if defined?(ErrorOccurrence) && ErrorOccurrence.table_exists?
|
|
88
|
+
ErrorOccurrence.joins(:error_log)
|
|
89
|
+
else
|
|
90
|
+
ErrorLog.all
|
|
91
|
+
end
|
|
92
|
+
relation = relation.where(logs => { application_id: application_id }) if application_id.present?
|
|
93
|
+
|
|
94
|
+
since = ->(time) { ErrorLog.sanitize_sql_array([ "SUM(CASE WHEN #{column} >= ? THEN 1 ELSE 0 END)", time ]) }
|
|
95
|
+
|
|
96
|
+
rows = relation
|
|
97
|
+
.where("#{column} >= ?", now.beginning_of_week)
|
|
98
|
+
.group("#{logs}.error_type", "#{logs}.platform")
|
|
99
|
+
.order(Arel.sql("COUNT(*) DESC"))
|
|
100
|
+
.limit(MAX_ANOMALY_PAIRS)
|
|
101
|
+
.pluck(Arel.sql("#{logs}.error_type"), Arel.sql("#{logs}.platform"), Arel.sql("COUNT(*)"),
|
|
102
|
+
Arel.sql(since.call(now.beginning_of_day)), Arel.sql(since.call(now.beginning_of_hour)))
|
|
103
|
+
|
|
104
|
+
rows.to_h do |error_type, platform, weekly, daily, hourly|
|
|
105
|
+
[ [ error_type, platform ], { weekly: weekly.to_i, daily: daily.to_i, hourly: hourly.to_i } ]
|
|
106
|
+
end
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
# { [error_type, platform, baseline_type] => ErrorBaseline }, the most recent
|
|
110
|
+
# row of each, in ONE query. Baseline rows accumulate (one per calculation
|
|
111
|
+
# period), so "latest" is resolved in SQL with a join on MAX(period_start)
|
|
112
|
+
# rather than by loading the history. The join form is portable to every
|
|
113
|
+
# adapter; a row-value IN is not.
|
|
114
|
+
def self.latest_baselines_for(pairs)
|
|
115
|
+
table = ErrorBaseline.table_name
|
|
116
|
+
types = pairs.map(&:first).compact.uniq
|
|
117
|
+
return {} if types.empty?
|
|
118
|
+
|
|
119
|
+
latest = ErrorBaseline.where(error_type: types, baseline_type: BASELINE_PRECEDENCE)
|
|
120
|
+
.group(:error_type, :platform, :baseline_type)
|
|
121
|
+
.select(:error_type, :platform, :baseline_type, "MAX(period_start) AS latest_period_start")
|
|
122
|
+
|
|
123
|
+
ErrorBaseline
|
|
124
|
+
.joins("INNER JOIN (#{latest.to_sql}) latest_baselines ON " \
|
|
125
|
+
"latest_baselines.error_type = #{table}.error_type AND " \
|
|
126
|
+
"latest_baselines.platform = #{table}.platform AND " \
|
|
127
|
+
"latest_baselines.baseline_type = #{table}.baseline_type AND " \
|
|
128
|
+
"latest_baselines.latest_period_start = #{table}.period_start")
|
|
129
|
+
.index_by { |baseline| [ baseline.error_type, baseline.platform, baseline.baseline_type ] }
|
|
130
|
+
end
|
|
131
|
+
private_class_method :current_counts_by_pair, :latest_baselines_for
|
|
132
|
+
|
|
26
133
|
def initialize(error_type, platform)
|
|
27
134
|
@error_type = error_type
|
|
28
135
|
@platform = platform
|
|
@@ -5,6 +5,9 @@ module RailsErrorDashboard
|
|
|
5
5
|
# Query: Fetch dashboard statistics
|
|
6
6
|
# This is a read operation that aggregates error data for the dashboard
|
|
7
7
|
class DashboardStats
|
|
8
|
+
# One instance answers ONE call: several aggregates (today's event count,
|
|
9
|
+
# the 7-day trend, spike detection) are memoised on it so that they are
|
|
10
|
+
# computed once per call rather than once per card that shows them.
|
|
8
11
|
def initialize(application_id: nil)
|
|
9
12
|
@application_id = application_id
|
|
10
13
|
end
|
|
@@ -24,7 +27,7 @@ module RailsErrorDashboard
|
|
|
24
27
|
# rows reported five users hitting one error as "1 error today".
|
|
25
28
|
# Summing is also exact during a storm: counted-only events never
|
|
26
29
|
# create occurrence rows, but they DO raise occurrence_count.
|
|
27
|
-
total_today:
|
|
30
|
+
total_today: today_event_count,
|
|
28
31
|
total_week: event_count_since(7.days.ago),
|
|
29
32
|
total_month: event_count_since(30.days.ago),
|
|
30
33
|
unresolved: base_scope.unresolved.count,
|
|
@@ -93,13 +96,14 @@ module RailsErrorDashboard
|
|
|
93
96
|
end
|
|
94
97
|
|
|
95
98
|
def cache_key
|
|
96
|
-
#
|
|
97
|
-
#
|
|
98
|
-
#
|
|
99
|
+
# The cache GENERATION, not maximum(:updated_at): the timestamp cost a
|
|
100
|
+
# query per key build and changed on every capture, so the cache never
|
|
101
|
+
# hit while errors were arriving. Freshness after a capture is the
|
|
102
|
+
# 1-minute TTL; user actions bump the generation (AnalyticsCacheManager).
|
|
99
103
|
[
|
|
100
104
|
"dashboard_stats",
|
|
101
105
|
@application_id || "all",
|
|
102
|
-
|
|
106
|
+
Services::AnalyticsCacheManager.generation,
|
|
103
107
|
Time.current.hour
|
|
104
108
|
].join("/")
|
|
105
109
|
end
|
|
@@ -147,7 +151,7 @@ module RailsErrorDashboard
|
|
|
147
151
|
recorded = occurrence_scope
|
|
148
152
|
.where("#{ErrorOccurrence.table_name}.occurred_at >= ?", Time.current.beginning_of_day)
|
|
149
153
|
.count
|
|
150
|
-
recorded <
|
|
154
|
+
recorded < today_event_count
|
|
151
155
|
rescue StandardError
|
|
152
156
|
false
|
|
153
157
|
end
|
|
@@ -169,9 +173,10 @@ module RailsErrorDashboard
|
|
|
169
173
|
|
|
170
174
|
# Get 7-day error trend (daily counts)
|
|
171
175
|
def errors_trend_7d
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
176
|
+
@errors_trend_7d ||=
|
|
177
|
+
base_scope.where("occurred_at >= ?", 7.days.ago)
|
|
178
|
+
.group_by_day(:occurred_at, range: 7.days.ago.to_date..Date.current, default_value: 0)
|
|
179
|
+
.sum(:occurrence_count)
|
|
175
180
|
end
|
|
176
181
|
|
|
177
182
|
# Get error counts by severity for last 7 days
|
|
@@ -193,21 +198,25 @@ module RailsErrorDashboard
|
|
|
193
198
|
|
|
194
199
|
# Detect if there's an error spike
|
|
195
200
|
# Uses baselines if available, falls back to simple 2x average
|
|
201
|
+
#
|
|
202
|
+
# Memoised: the stats hash asks twice (spike_detected and spike_info), and
|
|
203
|
+
# this runs from the live stats broadcast inside the capture path.
|
|
196
204
|
def spike_detected?
|
|
197
|
-
return
|
|
205
|
+
return @spike_detected if defined?(@spike_detected)
|
|
198
206
|
|
|
199
|
-
|
|
207
|
+
@spike_detected = compute_spike_detected
|
|
208
|
+
end
|
|
200
209
|
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
210
|
+
def compute_spike_detected
|
|
211
|
+
trend = errors_trend_7d
|
|
212
|
+
return false if trend.empty?
|
|
213
|
+
return true if baseline_anomalies.any?
|
|
205
214
|
|
|
206
215
|
# Fall back to simple 2x average detection
|
|
207
|
-
avg_count =
|
|
216
|
+
avg_count = trend.values.sum / 7.0
|
|
208
217
|
return false if avg_count.zero?
|
|
209
218
|
|
|
210
|
-
|
|
219
|
+
today_event_count >= (avg_count * 2)
|
|
211
220
|
end
|
|
212
221
|
|
|
213
222
|
# Get spike information
|
|
@@ -215,7 +224,7 @@ module RailsErrorDashboard
|
|
|
215
224
|
def spike_info
|
|
216
225
|
return nil unless spike_detected?
|
|
217
226
|
|
|
218
|
-
today_count =
|
|
227
|
+
today_count = today_event_count
|
|
219
228
|
avg_count = (errors_trend_7d.values.sum / 7.0).round(1)
|
|
220
229
|
|
|
221
230
|
info = {
|
|
@@ -226,46 +235,35 @@ module RailsErrorDashboard
|
|
|
226
235
|
}
|
|
227
236
|
|
|
228
237
|
# Add baseline info if available
|
|
229
|
-
baseline_info = baseline_anomaly_info
|
|
238
|
+
baseline_info = baseline_anomaly_info
|
|
230
239
|
info.merge!(baseline_info) if baseline_info.present?
|
|
231
240
|
|
|
232
241
|
info
|
|
233
242
|
end
|
|
234
243
|
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
244
|
+
def today_event_count
|
|
245
|
+
@today_event_count ||= event_count_since(Time.current.beginning_of_day)
|
|
246
|
+
end
|
|
247
|
+
|
|
248
|
+
# Every anomalous (error_type, platform) pair, from a fixed number of
|
|
249
|
+
# queries (BaselineStats.current_anomalies), loaded once per call. This
|
|
250
|
+
# used to be about six queries per distinct pair, run twice.
|
|
251
|
+
def baseline_anomalies
|
|
252
|
+
return @baseline_anomalies if defined?(@baseline_anomalies)
|
|
238
253
|
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
254
|
+
@baseline_anomalies = if defined?(Queries::BaselineStats)
|
|
255
|
+
Queries::BaselineStats.current_anomalies(sensitivity: 2, application_id: @application_id)
|
|
256
|
+
else
|
|
257
|
+
[]
|
|
243
258
|
end
|
|
244
259
|
end
|
|
245
260
|
|
|
246
261
|
# Get baseline anomaly information
|
|
247
|
-
def baseline_anomaly_info
|
|
248
|
-
return nil
|
|
249
|
-
|
|
250
|
-
# Find the most anomalous error type
|
|
251
|
-
anomalies = base_scope.distinct.pluck(:error_type, :platform).compact.map do |(error_type, platform)|
|
|
252
|
-
result = Queries::BaselineStats.new(error_type, platform)
|
|
253
|
-
.check_current_anomaly(sensitivity: 2, application_id: @application_id)
|
|
254
|
-
next unless result[:anomaly]
|
|
255
|
-
|
|
256
|
-
{
|
|
257
|
-
error_type: error_type,
|
|
258
|
-
platform: platform,
|
|
259
|
-
count: result[:current_count],
|
|
260
|
-
level: result[:level],
|
|
261
|
-
std_devs_above: result[:std_devs_above]
|
|
262
|
-
}
|
|
263
|
-
end.compact
|
|
264
|
-
|
|
265
|
-
return nil if anomalies.empty?
|
|
262
|
+
def baseline_anomaly_info
|
|
263
|
+
return nil if baseline_anomalies.empty?
|
|
266
264
|
|
|
267
265
|
# Return info about worst anomaly
|
|
268
|
-
worst =
|
|
266
|
+
worst = baseline_anomalies.max_by { |a| a[:std_devs_above] || 0 }
|
|
269
267
|
{
|
|
270
268
|
baseline_detected: true,
|
|
271
269
|
anomaly_error_type: worst[:error_type],
|
|
@@ -283,7 +281,7 @@ module RailsErrorDashboard
|
|
|
283
281
|
# make a real failure percentage from, so the honest figure is the rate
|
|
284
282
|
# itself, uncapped, labelled with its unit.
|
|
285
283
|
def error_rate
|
|
286
|
-
today_events =
|
|
284
|
+
today_events = today_event_count
|
|
287
285
|
return 0.0 if today_events.zero?
|
|
288
286
|
|
|
289
287
|
hours_today = ((Time.current - Time.current.beginning_of_day) / 1.hour).round(1)
|
|
@@ -299,11 +297,12 @@ module RailsErrorDashboard
|
|
|
299
297
|
# affected user. Storm count-only events create no occurrence row, so
|
|
300
298
|
# this is a floor during a storm -- affected_users_incomplete? says when.
|
|
301
299
|
def affected_users_today
|
|
302
|
-
distinct_affected_users(Time.current.beginning_of_day, nil)
|
|
300
|
+
@affected_users_today ||= distinct_affected_users(Time.current.beginning_of_day, nil)
|
|
303
301
|
end
|
|
304
302
|
|
|
305
303
|
def affected_users_yesterday
|
|
306
|
-
|
|
304
|
+
@affected_users_yesterday ||=
|
|
305
|
+
distinct_affected_users(1.day.ago.beginning_of_day, Time.current.beginning_of_day)
|
|
307
306
|
end
|
|
308
307
|
|
|
309
308
|
def distinct_affected_users(from, to)
|
|
@@ -334,7 +333,13 @@ module RailsErrorDashboard
|
|
|
334
333
|
|
|
335
334
|
# Calculate percentage change in errors (today vs yesterday)
|
|
336
335
|
def trend_percentage
|
|
337
|
-
|
|
336
|
+
return @trend_percentage if defined?(@trend_percentage)
|
|
337
|
+
|
|
338
|
+
@trend_percentage = compute_trend_percentage
|
|
339
|
+
end
|
|
340
|
+
|
|
341
|
+
def compute_trend_percentage
|
|
342
|
+
today = today_event_count
|
|
338
343
|
yesterday = event_count_between(1.day.ago.beginning_of_day, Time.current.beginning_of_day)
|
|
339
344
|
|
|
340
345
|
return 0.0 if today.zero? && yesterday.zero?
|
|
@@ -224,12 +224,15 @@ module RailsErrorDashboard
|
|
|
224
224
|
previous_start = @start_date
|
|
225
225
|
previous_end = current_start
|
|
226
226
|
|
|
227
|
-
|
|
227
|
+
# Both periods come from base_query, which carries the application
|
|
228
|
+
# filter (and the window start). Querying ErrorLog directly made this
|
|
229
|
+
# the one panel on an application-filtered page that counted every app.
|
|
230
|
+
current_errors = base_query
|
|
228
231
|
.where("occurred_at >= ?", current_start)
|
|
229
232
|
.count
|
|
230
233
|
|
|
231
|
-
previous_errors =
|
|
232
|
-
.where("occurred_at
|
|
234
|
+
previous_errors = base_query
|
|
235
|
+
.where("occurred_at < ?", previous_end)
|
|
233
236
|
.count
|
|
234
237
|
|
|
235
238
|
change_percentage = if previous_errors > 0
|
|
@@ -5,6 +5,9 @@ module RailsErrorDashboard
|
|
|
5
5
|
# Query: Fetch errors with filtering and pagination
|
|
6
6
|
# This is a read operation that returns a filtered collection of errors
|
|
7
7
|
class ErrorsList
|
|
8
|
+
# Escape character for LIKE patterns; see filter_by_search.
|
|
9
|
+
LIKE_ESCAPE = "!"
|
|
10
|
+
|
|
8
11
|
def self.call(filters = {})
|
|
9
12
|
new(filters).call
|
|
10
13
|
end
|
|
@@ -129,9 +132,19 @@ module RailsErrorDashboard
|
|
|
129
132
|
else
|
|
130
133
|
# Fall back to LIKE for SQLite/MySQL - search across all relevant fields
|
|
131
134
|
# Use LOWER() for case-insensitive search
|
|
132
|
-
|
|
135
|
+
#
|
|
136
|
+
# % and _ in the term are escaped so they match themselves: a search
|
|
137
|
+
# for "snake_case" must not match "snakeXcase", and "%" must not
|
|
138
|
+
# match every row. The escape character is named explicitly because
|
|
139
|
+
# SQLite has no default one, and it is "!" rather than a backslash
|
|
140
|
+
# because a backslash inside a quoted SQL literal means different
|
|
141
|
+
# things to MySQL and to everyone else.
|
|
142
|
+
escaped = ActiveRecord::Base.sanitize_sql_like(@filters[:search].to_s, LIKE_ESCAPE)
|
|
143
|
+
search_pattern = "%#{escaped}%"
|
|
133
144
|
query.where(
|
|
134
|
-
"LOWER(message) LIKE LOWER(?)
|
|
145
|
+
"LOWER(message) LIKE LOWER(?) ESCAPE '#{LIKE_ESCAPE}' " \
|
|
146
|
+
"OR LOWER(COALESCE(backtrace, '')) LIKE LOWER(?) ESCAPE '#{LIKE_ESCAPE}' " \
|
|
147
|
+
"OR LOWER(error_type) LIKE LOWER(?) ESCAPE '#{LIKE_ESCAPE}'",
|
|
135
148
|
search_pattern, search_pattern, search_pattern
|
|
136
149
|
)
|
|
137
150
|
end
|
|
@@ -78,7 +78,7 @@ module RailsErrorDashboard
|
|
|
78
78
|
if error_prefix.present?
|
|
79
79
|
candidates += ErrorLog
|
|
80
80
|
.where(platform: target_error.platform)
|
|
81
|
-
.where("error_type LIKE ?", "%#{error_prefix}%")
|
|
81
|
+
.where("error_type LIKE ? ESCAPE '!'", "%#{ActiveRecord::Base.sanitize_sql_like(error_prefix, "!")}%")
|
|
82
82
|
.where.not(id: target_error.id)
|
|
83
83
|
.where.not(id: candidates.map(&:id))
|
|
84
84
|
.limit(20)
|