rails_error_dashboard 0.12.1 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/app/controllers/rails_error_dashboard/application_controller.rb +92 -11
- data/app/controllers/rails_error_dashboard/errors_controller.rb +111 -34
- data/app/controllers/rails_error_dashboard/webhooks_controller.rb +3 -0
- data/app/helpers/rails_error_dashboard/application_helper.rb +46 -1
- data/app/helpers/rails_error_dashboard/backtrace_helper.rb +8 -0
- data/app/jobs/rails_error_dashboard/async_error_logging_job.rb +10 -0
- data/app/jobs/rails_error_dashboard/concerns/plain_channel_message.rb +71 -0
- data/app/jobs/rails_error_dashboard/notification_burst_summary_job.rb +55 -0
- data/app/jobs/rails_error_dashboard/retention_cleanup_job.rb +122 -1
- data/app/jobs/rails_error_dashboard/storm_flush_job.rb +7 -4
- data/app/jobs/rails_error_dashboard/storm_notification_job.rb +8 -44
- data/app/models/rails_error_dashboard/error_baseline.rb +12 -9
- data/app/models/rails_error_dashboard/error_comment.rb +0 -5
- data/app/models/rails_error_dashboard/error_log.rb +19 -3
- data/app/models/rails_error_dashboard/error_logs_record.rb +34 -0
- data/app/models/rails_error_dashboard/error_occurrence.rb +9 -1
- data/app/models/rails_error_dashboard/event_count.rb +132 -0
- data/app/models/rails_error_dashboard/event_timing_gap.rb +55 -0
- data/app/views/layouts/rails_error_dashboard.html.erb +58 -5
- data/app/views/rails_error_dashboard/errors/_discussion.html.erb +18 -11
- data/app/views/rails_error_dashboard/errors/_issue_section.html.erb +22 -8
- data/app/views/rails_error_dashboard/errors/_request_context.html.erb +2 -0
- data/app/views/rails_error_dashboard/errors/_sidebar_metadata.html.erb +15 -6
- data/app/views/rails_error_dashboard/errors/analytics.html.erb +8 -8
- data/app/views/rails_error_dashboard/errors/diagnostic_dumps.html.erb +1 -1
- data/app/views/rails_error_dashboard/errors/index.html.erb +6 -1
- data/app/views/rails_error_dashboard/errors/overview.html.erb +12 -0
- data/app/views/rails_error_dashboard/errors/platform_comparison.html.erb +7 -7
- data/app/views/rails_error_dashboard/errors/releases.html.erb +2 -2
- data/app/views/rails_error_dashboard/errors/settings.html.erb +2 -0
- data/app/views/rails_error_dashboard/errors/show.html.erb +3 -3
- data/config/locales/de.yml +30 -0
- data/config/locales/en.yml +37 -0
- data/config/locales/es.yml +30 -0
- data/config/locales/fr.yml +30 -0
- data/config/locales/it.yml +30 -0
- data/config/locales/ja.yml +30 -0
- data/config/locales/pl.yml +30 -0
- data/config/locales/pt-BR.yml +30 -0
- data/config/locales/ru.yml +30 -0
- data/config/locales/uk.yml +30 -0
- data/config/locales/zh-CN.yml +30 -0
- data/db/migrate/20260917000001_add_last_notified_at_to_error_logs.rb +40 -0
- data/db/migrate/20260919000001_create_event_counts.rb +71 -0
- data/db/migrate/20260920000001_add_buckets_incomplete_to_storm_events.rb +25 -0
- data/db/migrate/20260920000002_create_event_timing_gaps.rb +55 -0
- data/lib/generators/rails_error_dashboard/install/templates/initializer.rb +12 -1
- data/lib/rails_error_dashboard/commands/assign_error.rb +12 -2
- data/lib/rails_error_dashboard/commands/backfill_environments.rb +2 -0
- data/lib/rails_error_dashboard/commands/backfill_resolved_at.rb +42 -0
- data/lib/rails_error_dashboard/commands/batch_delete_errors.rb +1 -0
- data/lib/rails_error_dashboard/commands/batch_mute_errors.rb +2 -0
- data/lib/rails_error_dashboard/commands/batch_resolve_errors.rb +2 -0
- data/lib/rails_error_dashboard/commands/batch_unmute_errors.rb +2 -0
- data/lib/rails_error_dashboard/commands/find_or_increment_error.rb +109 -11
- data/lib/rails_error_dashboard/commands/flush_rack_attack_events.rb +3 -1
- data/lib/rails_error_dashboard/commands/flush_storm_counts.rb +252 -12
- data/lib/rails_error_dashboard/commands/flush_swallowed_exceptions.rb +5 -2
- data/lib/rails_error_dashboard/commands/link_existing_issue.rb +1 -0
- data/lib/rails_error_dashboard/commands/log_error.rb +291 -40
- data/lib/rails_error_dashboard/commands/mute_error.rb +1 -0
- data/lib/rails_error_dashboard/commands/resolve_error.rb +2 -0
- data/lib/rails_error_dashboard/commands/scrub_invalid_encoding.rb +104 -0
- data/lib/rails_error_dashboard/commands/snooze_error.rb +34 -10
- data/lib/rails_error_dashboard/commands/unmute_error.rb +1 -0
- data/lib/rails_error_dashboard/commands/update_error_priority.rb +23 -2
- data/lib/rails_error_dashboard/commands/update_error_status.rb +33 -6
- data/lib/rails_error_dashboard/configuration.rb +39 -1
- data/lib/rails_error_dashboard/engine.rb +28 -0
- data/lib/rails_error_dashboard/manual_error_reporter.rb +16 -5
- data/lib/rails_error_dashboard/queries/analytics_stats.rb +89 -30
- data/lib/rails_error_dashboard/queries/baseline_stats.rb +107 -0
- data/lib/rails_error_dashboard/queries/dashboard_stats.rb +216 -74
- data/lib/rails_error_dashboard/queries/error_correlation.rb +6 -3
- data/lib/rails_error_dashboard/queries/errors_list.rb +15 -2
- data/lib/rails_error_dashboard/queries/event_volume.rb +503 -0
- data/lib/rails_error_dashboard/queries/similar_errors.rb +1 -1
- data/lib/rails_error_dashboard/services/analytics_cache_manager.rb +43 -18
- data/lib/rails_error_dashboard/services/backtrace_processor.rb +3 -1
- data/lib/rails_error_dashboard/services/breadcrumb_collector.rb +23 -0
- data/lib/rails_error_dashboard/services/cause_chain_extractor.rb +3 -1
- data/lib/rails_error_dashboard/services/codeberg_issue_client.rb +13 -2
- data/lib/rails_error_dashboard/services/diagnostic_dump_generator.rb +5 -3
- data/lib/rails_error_dashboard/services/encoding_sanitizer.rb +80 -0
- data/lib/rails_error_dashboard/services/error_broadcaster.rb +180 -32
- data/lib/rails_error_dashboard/services/error_hash_generator.rb +9 -3
- data/lib/rails_error_dashboard/services/error_notification_dispatcher.rb +13 -0
- data/lib/rails_error_dashboard/services/exception_filter.rb +55 -0
- data/lib/rails_error_dashboard/services/git_head_reader.rb +102 -0
- data/lib/rails_error_dashboard/services/notification_throttler.rb +172 -28
- data/lib/rails_error_dashboard/services/sensitive_data_filter.rb +47 -1
- data/lib/rails_error_dashboard/services/storm_protection/circuit_breaker.rb +59 -3
- data/lib/rails_error_dashboard/services/storm_protection/count_buffer.rb +64 -6
- data/lib/rails_error_dashboard/services/storm_protection/fingerprint_buckets.rb +25 -2
- data/lib/rails_error_dashboard/services/storm_protection/gate.rb +32 -5
- data/lib/rails_error_dashboard/services/swallowed_exception_tracker.rb +99 -23
- data/lib/rails_error_dashboard/services/url_safety.rb +40 -0
- data/lib/rails_error_dashboard/services/variable_serializer.rb +125 -10
- data/lib/rails_error_dashboard/subscribers/breadcrumb_subscriber.rb +111 -0
- data/lib/rails_error_dashboard/subscribers/issue_tracker_subscriber.rb +11 -2
- data/lib/rails_error_dashboard/value_objects/error_context.rb +40 -3
- data/lib/rails_error_dashboard/version.rb +1 -1
- data/lib/rails_error_dashboard.rb +34 -0
- data/lib/tasks/error_dashboard.rake +54 -4
- metadata +16 -2
|
@@ -4,7 +4,14 @@ module RailsErrorDashboard
|
|
|
4
4
|
module Commands
|
|
5
5
|
# Command: Snooze an error for a given number of hours
|
|
6
6
|
# This is a write operation that sets snoozed_until and optionally creates a comment
|
|
7
|
+
# Returns {success: bool, error: ErrorLog}; a failure also carries
|
|
8
|
+
# reason: :invalid_hours and writes nothing.
|
|
7
9
|
class SnoozeError
|
|
10
|
+
# 30 days. The form offers at most a week; the cap is for anything that
|
|
11
|
+
# does not come from the form. Without one, a negative value "snoozed"
|
|
12
|
+
# into the past and a huge one overflowed the timestamp.
|
|
13
|
+
MAX_SNOOZE_HOURS = 720
|
|
14
|
+
|
|
8
15
|
def self.call(error_id, hours:, reason: nil)
|
|
9
16
|
new(error_id, hours, reason).call
|
|
10
17
|
end
|
|
@@ -17,18 +24,35 @@ module RailsErrorDashboard
|
|
|
17
24
|
|
|
18
25
|
def call
|
|
19
26
|
error = ErrorLog.find(@error_id)
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
27
|
+
hours = whole_hours(@hours)
|
|
28
|
+
|
|
29
|
+
unless hours && (1..MAX_SNOOZE_HOURS).cover?(hours)
|
|
30
|
+
return { success: false, error: error, reason: :invalid_hours }
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
error.transaction do
|
|
34
|
+
if @reason.present?
|
|
35
|
+
error.comments.create!(
|
|
36
|
+
author_name: error.assigned_to || "System",
|
|
37
|
+
body: "Snoozed for #{hours} hours: #{@reason}"
|
|
38
|
+
)
|
|
39
|
+
end
|
|
40
|
+
|
|
41
|
+
error.update!(snoozed_until: hours.hours.from_now)
|
|
28
42
|
end
|
|
29
43
|
|
|
30
|
-
error
|
|
31
|
-
|
|
44
|
+
{ success: true, error: error }
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
private
|
|
48
|
+
|
|
49
|
+
# An Integer, or a String that is one ("24"). Anything else -- a Float, a
|
|
50
|
+
# nested parameter, "abc" -- is nil rather than a guess.
|
|
51
|
+
def whole_hours(value)
|
|
52
|
+
case value
|
|
53
|
+
when Integer then value
|
|
54
|
+
when String then Integer(value.strip, 10, exception: false)
|
|
55
|
+
end
|
|
32
56
|
end
|
|
33
57
|
end
|
|
34
58
|
end
|
|
@@ -4,6 +4,8 @@ module RailsErrorDashboard
|
|
|
4
4
|
module Commands
|
|
5
5
|
# Command: Update the priority level of an error
|
|
6
6
|
# This is a write operation that updates the priority_level field on an ErrorLog record
|
|
7
|
+
# Returns {success: bool, error: ErrorLog}; a failure also carries
|
|
8
|
+
# reason: :invalid_priority and leaves the existing priority alone.
|
|
7
9
|
class UpdateErrorPriority
|
|
8
10
|
def self.call(error_id, priority_level:)
|
|
9
11
|
new(error_id, priority_level).call
|
|
@@ -16,8 +18,27 @@ module RailsErrorDashboard
|
|
|
16
18
|
|
|
17
19
|
def call
|
|
18
20
|
error = ErrorLog.find(@error_id)
|
|
19
|
-
|
|
20
|
-
|
|
21
|
+
level = whole_number(@priority_level)
|
|
22
|
+
|
|
23
|
+
# The column is an integer, so an unchecked "x" was cast and stored as
|
|
24
|
+
# 0 -- silently replacing a real priority with Low.
|
|
25
|
+
unless ErrorLog::PRIORITY_LEVELS.key?(level)
|
|
26
|
+
return { success: false, error: error, reason: :invalid_priority }
|
|
27
|
+
end
|
|
28
|
+
|
|
29
|
+
error.update!(priority_level: level)
|
|
30
|
+
{ success: true, error: error }
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
private
|
|
34
|
+
|
|
35
|
+
# An Integer, or a String that is one ("3"). A nested parameter, a Float
|
|
36
|
+
# or free text is nil.
|
|
37
|
+
def whole_number(value)
|
|
38
|
+
case value
|
|
39
|
+
when Integer then value
|
|
40
|
+
when String then Integer(value.strip, 10, exception: false)
|
|
41
|
+
end
|
|
21
42
|
end
|
|
22
43
|
end
|
|
23
44
|
end
|
|
@@ -4,7 +4,8 @@ module RailsErrorDashboard
|
|
|
4
4
|
module Commands
|
|
5
5
|
# Command: Update the status of an error with optional comment
|
|
6
6
|
# This is a write operation that validates transitions and updates status
|
|
7
|
-
# Returns {success: bool, error: ErrorLog}
|
|
7
|
+
# Returns {success: bool, error: ErrorLog}; a failure also carries
|
|
8
|
+
# reason: :unknown_status or :invalid_transition so the caller can say which.
|
|
8
9
|
class UpdateErrorStatus
|
|
9
10
|
def self.call(error_id, status:, comment: nil)
|
|
10
11
|
new(error_id, status, comment).call
|
|
@@ -19,15 +20,17 @@ module RailsErrorDashboard
|
|
|
19
20
|
def call
|
|
20
21
|
error = ErrorLog.find(@error_id)
|
|
21
22
|
|
|
23
|
+
# A nested param arrives as a Hash-like object, never a known status.
|
|
24
|
+
unless @status.is_a?(String) && ErrorLog::STATUSES.include?(@status)
|
|
25
|
+
return { success: false, error: error, reason: :unknown_status }
|
|
26
|
+
end
|
|
27
|
+
|
|
22
28
|
unless error.can_transition_to?(@status)
|
|
23
|
-
return { success: false, error: error }
|
|
29
|
+
return { success: false, error: error, reason: :invalid_transition }
|
|
24
30
|
end
|
|
25
31
|
|
|
26
32
|
error.transaction do
|
|
27
|
-
error.update!(
|
|
28
|
-
|
|
29
|
-
# Auto-resolve if status is "resolved"
|
|
30
|
-
error.update!(resolved: true) if @status == "resolved"
|
|
33
|
+
error.update!(status_attributes(error))
|
|
31
34
|
|
|
32
35
|
# Add comment about status change
|
|
33
36
|
if @comment.present?
|
|
@@ -38,8 +41,32 @@ module RailsErrorDashboard
|
|
|
38
41
|
end
|
|
39
42
|
end
|
|
40
43
|
|
|
44
|
+
# The stat cards are cached; a user action must show up at once.
|
|
45
|
+
Services::AnalyticsCacheManager.clear
|
|
46
|
+
|
|
41
47
|
{ success: true, error: error }
|
|
42
48
|
end
|
|
49
|
+
|
|
50
|
+
private
|
|
51
|
+
|
|
52
|
+
# One write, so the three columns can never disagree. resolved_at is what
|
|
53
|
+
# MTTR is computed from: leaving it nil (as this command used to) dropped
|
|
54
|
+
# every error resolved through the status workflow from the MTTR figures,
|
|
55
|
+
# and leaving it set on a reopened error kept a stale resolution time.
|
|
56
|
+
# Only "resolved" sets the flag -- wont_fix stays resolved: false.
|
|
57
|
+
def status_attributes(error)
|
|
58
|
+
attrs = { status: @status }
|
|
59
|
+
|
|
60
|
+
if @status == "resolved"
|
|
61
|
+
attrs[:resolved] = true
|
|
62
|
+
attrs[:resolved_at] = Time.current
|
|
63
|
+
elsif error.status == "resolved" || error.resolved?
|
|
64
|
+
attrs[:resolved] = false
|
|
65
|
+
attrs[:resolved_at] = nil
|
|
66
|
+
end
|
|
67
|
+
|
|
68
|
+
attrs
|
|
69
|
+
end
|
|
43
70
|
end
|
|
44
71
|
end
|
|
45
72
|
end
|
|
@@ -157,6 +157,8 @@ module RailsErrorDashboard
|
|
|
157
157
|
attr_accessor :notification_minimum_severity # Minimum severity to notify (default: :low = notify all)
|
|
158
158
|
attr_accessor :notification_cooldown_minutes # Per-error cooldown in minutes (default: 5, 0 = disabled)
|
|
159
159
|
attr_accessor :notification_threshold_alerts # Occurrence milestones that trigger notification (default: [10, 50, 100, 500, 1000])
|
|
160
|
+
attr_accessor :notification_burst_limit # Max FIRST-OCCURRENCE notifications per window, per process (default: 10, 0 = no cap)
|
|
161
|
+
attr_accessor :notification_burst_window_seconds # Length of that window in seconds (default: 60)
|
|
160
162
|
|
|
161
163
|
# Breadcrumbs (request activity trail)
|
|
162
164
|
attr_accessor :enable_breadcrumbs # Master switch (default: false)
|
|
@@ -180,6 +182,12 @@ module RailsErrorDashboard
|
|
|
180
182
|
attr_accessor :local_variable_max_array_items # Max array items to serialize (default: 10)
|
|
181
183
|
attr_accessor :local_variable_max_hash_items # Max hash entries to serialize (default: 20)
|
|
182
184
|
attr_accessor :local_variable_filter_patterns # Additional sensitive name patterns (default: [])
|
|
185
|
+
# Calling #inspect on an unknown object runs arbitrary APPLICATION code on
|
|
186
|
+
# the failure path; truncating the result bounds storage, not cost. The
|
|
187
|
+
# default is a safe structural summary, with full inspect per type by
|
|
188
|
+
# opt-in and a wall-clock budget even then.
|
|
189
|
+
attr_accessor :local_variable_inspect_allowlist # Class names whose #inspect may run (default: safe built-ins)
|
|
190
|
+
attr_accessor :local_variable_inspect_budget_ms # Wall-clock budget for one #inspect (default: 5)
|
|
183
191
|
|
|
184
192
|
# Instance variable capture from tp.self (receiver object at raise time)
|
|
185
193
|
attr_accessor :enable_instance_variables # Master switch (default: false)
|
|
@@ -302,7 +310,8 @@ module RailsErrorDashboard
|
|
|
302
310
|
|
|
303
311
|
@use_separate_database = ENV.fetch("USE_SEPARATE_ERROR_DB", "false") == "true"
|
|
304
312
|
|
|
305
|
-
# Retention policy - days
|
|
313
|
+
# Retention policy - days an error may go unseen (last_seen_at) before it is
|
|
314
|
+
# deleted automatically (default: 90). An error still occurring is kept.
|
|
306
315
|
# Set to nil to keep errors forever (not recommended for production)
|
|
307
316
|
# Schedule cleanup: RailsErrorDashboard::RetentionCleanupJob.perform_later
|
|
308
317
|
@retention_days = 90
|
|
@@ -384,6 +393,8 @@ module RailsErrorDashboard
|
|
|
384
393
|
@notification_minimum_severity = :low # Notify on all severities (current behavior)
|
|
385
394
|
@notification_cooldown_minutes = 5 # 5 min cooldown per error_hash (0 = disabled)
|
|
386
395
|
@notification_threshold_alerts = [ 10, 50, 100, 500, 1000 ] # Occurrence milestones
|
|
396
|
+
@notification_burst_limit = 10 # New-error notifications per window, per process (0 = no cap)
|
|
397
|
+
@notification_burst_window_seconds = 60 # One summary message replaces the rest of the window
|
|
387
398
|
|
|
388
399
|
# Breadcrumbs defaults - OFF by default (opt-in)
|
|
389
400
|
@enable_breadcrumbs = false # Master switch
|
|
@@ -410,6 +421,20 @@ module RailsErrorDashboard
|
|
|
410
421
|
@local_variable_max_array_items = 10 # Max array items to serialize
|
|
411
422
|
@local_variable_max_hash_items = 20 # Max hash entries to serialize
|
|
412
423
|
@local_variable_filter_patterns = [] # Additional sensitive variable name patterns
|
|
424
|
+
# Struct is serialized MEMBER-WISE (never via its own #inspect), so it
|
|
425
|
+
# needs no allowlist entry -- see VariableSerializer.serialize_struct.
|
|
426
|
+
#
|
|
427
|
+
# ActiveModel is deliberately NOT allowlisted. Reading
|
|
428
|
+
# ActiveModel::Attributes#attributes runs each attribute's type cast,
|
|
429
|
+
# which is application code, so neither #inspect nor member-wise reading
|
|
430
|
+
# can be bounded for it; it gets a safe summary instead.
|
|
431
|
+
#
|
|
432
|
+
# Anything added here opts that type IN to unbounded execution: the only
|
|
433
|
+
# way to interrupt arbitrary Ruby mid-call is Timeout, which is not safe
|
|
434
|
+
# on the capture path. The budget below selects the stored OUTPUT after
|
|
435
|
+
# the fact; it does not bound the work.
|
|
436
|
+
@local_variable_inspect_allowlist = []
|
|
437
|
+
@local_variable_inspect_budget_ms = 5
|
|
413
438
|
|
|
414
439
|
# Instance variable capture defaults - OFF by default (opt-in)
|
|
415
440
|
@enable_instance_variables = false # Capture ivars from tp.self at raise time
|
|
@@ -854,6 +879,19 @@ module RailsErrorDashboard
|
|
|
854
879
|
errors << "notification_threshold_alerts must be an Array (got: #{notification_threshold_alerts.class})"
|
|
855
880
|
end
|
|
856
881
|
|
|
882
|
+
# Validate the first-occurrence burst cap (non-negative integers; 0 or nil
|
|
883
|
+
# turns the cap off)
|
|
884
|
+
{
|
|
885
|
+
notification_burst_limit: notification_burst_limit,
|
|
886
|
+
notification_burst_window_seconds: notification_burst_window_seconds
|
|
887
|
+
}.each do |name, value|
|
|
888
|
+
next if value.nil?
|
|
889
|
+
|
|
890
|
+
unless value.is_a?(Integer) && value >= 0
|
|
891
|
+
errors << "#{name} must be a non-negative Integer (got: #{value.inspect})"
|
|
892
|
+
end
|
|
893
|
+
end
|
|
894
|
+
|
|
857
895
|
# Log warnings (non-fatal issues)
|
|
858
896
|
warnings.each do |warning|
|
|
859
897
|
Rails.logger.warn "[Rails Error Dashboard] #{warning}" if defined?(Rails) && Rails.respond_to?(:logger) && Rails.logger
|
|
@@ -71,6 +71,11 @@ module RailsErrorDashboard
|
|
|
71
71
|
next
|
|
72
72
|
end
|
|
73
73
|
|
|
74
|
+
# Resolve the running commit now (three small file reads, memoised, never
|
|
75
|
+
# raises) so that no capture ever pays for it. Skipped when the SHA is
|
|
76
|
+
# configured: then it is never consulted.
|
|
77
|
+
RailsErrorDashboard.detected_git_sha if RailsErrorDashboard.configuration.git_sha.blank?
|
|
78
|
+
|
|
74
79
|
if RailsErrorDashboard.configuration.enable_error_subscriber
|
|
75
80
|
Rails.error.subscribe(RailsErrorDashboard::ErrorReporter.new)
|
|
76
81
|
end
|
|
@@ -80,6 +85,19 @@ module RailsErrorDashboard
|
|
|
80
85
|
RailsErrorDashboard::Subscribers::BreadcrumbSubscriber.subscribe!
|
|
81
86
|
end
|
|
82
87
|
|
|
88
|
+
# Give background jobs a breadcrumb buffer of their own. init_buffer had
|
|
89
|
+
# exactly one caller -- the Rack middleware -- so a job never entered the
|
|
90
|
+
# HTTP stack and had no buffer at all; every subscriber early-returns on
|
|
91
|
+
# `unless current_buffer`, so a failing job's SQL and custom crumbs were
|
|
92
|
+
# dropped in the one place an error is hardest to reproduce.
|
|
93
|
+
#
|
|
94
|
+
# Registered unconditionally and gated at PERFORM time, not here:
|
|
95
|
+
# enable_breadcrumbs defaults to false, and the ActiveJob callback list
|
|
96
|
+
# is fixed once the class loads, so a boot-time gate would leave this
|
|
97
|
+
# permanently unregistered for any host that turns the feature on in an
|
|
98
|
+
# initializer that runs later. Off, it costs one config read per job.
|
|
99
|
+
RailsErrorDashboard::Subscribers::BreadcrumbSubscriber.install_job_buffer!
|
|
100
|
+
|
|
83
101
|
# Subscribe to Rack Attack AS::Notifications events (requires Rack::Attack).
|
|
84
102
|
# Breadcrumbs are NOT required — events persist to their own table (issue #143).
|
|
85
103
|
if RailsErrorDashboard.configuration.enable_rack_attack_tracking &&
|
|
@@ -188,6 +206,16 @@ module RailsErrorDashboard
|
|
|
188
206
|
# Enable TracePoint(:raise) + TracePoint(:rescue) for swallowed exception detection
|
|
189
207
|
if RailsErrorDashboard.configuration.detect_swallowed_exceptions
|
|
190
208
|
RailsErrorDashboard::Services::SwallowedExceptionTracker.enable!
|
|
209
|
+
|
|
210
|
+
# Drain buffered counts at the end of every request and job. Without
|
|
211
|
+
# this the buffer is only ever drained by a LATER rescue on the SAME
|
|
212
|
+
# thread, so a swallowed exception that happens once stays invisible
|
|
213
|
+
# until the process exits. to_complete fires after the response body
|
|
214
|
+
# is closed, so it never delays a request (safety rule 2); the flush is
|
|
215
|
+
# deadline-gated, so a flood is still one write per interval.
|
|
216
|
+
Rails.application.executor.to_complete do
|
|
217
|
+
RailsErrorDashboard::Services::SwallowedExceptionTracker.flush_if_due!
|
|
218
|
+
end
|
|
191
219
|
end
|
|
192
220
|
|
|
193
221
|
# Import crash files from previous process death, then register at_exit hook
|
|
@@ -42,7 +42,11 @@ module RailsErrorDashboard
|
|
|
42
42
|
# @param app_version [String, nil] Version of the app where error occurred
|
|
43
43
|
# @param metadata [Hash, nil] Additional custom metadata about the error
|
|
44
44
|
# @param occurred_at [Time, nil] When the error occurred (defaults to Time.current)
|
|
45
|
-
# @param severity [Symbol, nil]
|
|
45
|
+
# @param severity [Symbol, nil] IGNORED, and kept only so existing callers
|
|
46
|
+
# do not break. Severity is not stored: ErrorLog#severity is CLASSIFIED
|
|
47
|
+
# from error_type by Services::SeverityClassifier, so the way to control
|
|
48
|
+
# it is the error_type you report. Accepting this silently made callers
|
|
49
|
+
# believe they had set a value that was never read.
|
|
46
50
|
# @param source [String, nil] Source identifier (e.g., "frontend", "mobile_app")
|
|
47
51
|
#
|
|
48
52
|
# @return [ErrorLog, nil] The created error log record, or nil if filtered/ignored
|
|
@@ -65,8 +69,7 @@ module RailsErrorDashboard
|
|
|
65
69
|
# user_agent: request.user_agent,
|
|
66
70
|
# ip_address: request.remote_ip,
|
|
67
71
|
# app_version: "1.2.3",
|
|
68
|
-
# metadata: { card_type: "visa", amount: 99.99 }
|
|
69
|
-
# severity: :high
|
|
72
|
+
# metadata: { card_type: "visa", amount: 99.99 }
|
|
70
73
|
# )
|
|
71
74
|
def self.report(
|
|
72
75
|
error_type:,
|
|
@@ -100,10 +103,18 @@ module RailsErrorDashboard
|
|
|
100
103
|
platform: platform,
|
|
101
104
|
app_version: app_version,
|
|
102
105
|
metadata: metadata,
|
|
103
|
-
occurred_at: occurred_at || Time.current
|
|
104
|
-
severity: severity
|
|
106
|
+
occurred_at: occurred_at || Time.current
|
|
105
107
|
}.compact # Remove nil values
|
|
106
108
|
|
|
109
|
+
# severity is deliberately NOT forwarded: nothing downstream reads it,
|
|
110
|
+
# and passing it on would keep the illusion that it does.
|
|
111
|
+
if severity.present?
|
|
112
|
+
RailsErrorDashboard::Logger.debug(
|
|
113
|
+
"[RailsErrorDashboard] ManualErrorReporter: severity: is ignored — " \
|
|
114
|
+
"severity is classified from error_type (#{error_type})."
|
|
115
|
+
)
|
|
116
|
+
end
|
|
117
|
+
|
|
107
118
|
# Use the existing LogError command
|
|
108
119
|
Commands::LogError.call(synthetic_exception, context)
|
|
109
120
|
end
|
|
@@ -17,7 +17,7 @@ module RailsErrorDashboard
|
|
|
17
17
|
|
|
18
18
|
def call
|
|
19
19
|
# Cache analytics data for 5 minutes to reduce database load
|
|
20
|
-
# Cache key includes days parameter and
|
|
20
|
+
# Cache key includes the days parameter and the cache generation
|
|
21
21
|
Rails.cache.fetch(cache_key, expires_in: 5.minutes) do
|
|
22
22
|
{
|
|
23
23
|
days: @days,
|
|
@@ -41,13 +41,14 @@ module RailsErrorDashboard
|
|
|
41
41
|
# - Query class name
|
|
42
42
|
# - Days parameter (different time ranges = different caches)
|
|
43
43
|
# - Application ID (per-app caching)
|
|
44
|
-
# -
|
|
44
|
+
# - The cache generation (bumped by user actions; see AnalyticsCacheManager).
|
|
45
|
+
# Captures do not bump it: they rely on the 5-minute TTL.
|
|
45
46
|
# - Start date (ensures correct time window)
|
|
46
47
|
[
|
|
47
48
|
"analytics_stats",
|
|
48
49
|
@days,
|
|
49
50
|
@application_id || "all",
|
|
50
|
-
|
|
51
|
+
Services::AnalyticsCacheManager.generation,
|
|
51
52
|
@start_date.to_date.to_s
|
|
52
53
|
].join("/")
|
|
53
54
|
end
|
|
@@ -66,43 +67,62 @@ module RailsErrorDashboard
|
|
|
66
67
|
|
|
67
68
|
# Two counting units, kept distinct on purpose.
|
|
68
69
|
#
|
|
69
|
-
# An ErrorLog row is a GROUP; its
|
|
70
|
-
#
|
|
71
|
-
#
|
|
72
|
-
#
|
|
73
|
-
#
|
|
70
|
+
# An ErrorLog row is a GROUP; its occurred_at is FIRST-SEEN and is never
|
|
71
|
+
# rewritten when the error recurs.
|
|
72
|
+
#
|
|
73
|
+
# EVENT figures -> Queries::EventVolume, which counts occurrence rows
|
|
74
|
+
# + storm buckets + the untracked remainder, each
|
|
75
|
+
# against its OWN timestamp.
|
|
76
|
+
# GROUP figures -> base_query, which selects groups by first-seen.
|
|
77
|
+
# Correct here: a group is the thing that gets
|
|
78
|
+
# resolved, and an event cannot be.
|
|
79
|
+
#
|
|
80
|
+
# Mixing them is what made this page disagree with the Overview: filtering
|
|
81
|
+
# GROUPS by first-seen and then summing their LIFETIME occurrence_count
|
|
82
|
+
# answers "how much total volume do the groups born in this window carry",
|
|
83
|
+
# not "how many events happened in this window". A group first seen in
|
|
84
|
+
# August that recurred today was excluded entirely, so the page reported
|
|
85
|
+
# zero events while its own affected-users table listed the very event.
|
|
74
86
|
def error_statistics
|
|
75
87
|
{
|
|
76
88
|
total: event_count,
|
|
77
89
|
total_groups: base_query.count,
|
|
78
90
|
unresolved: base_query.unresolved.count,
|
|
79
91
|
resolved: base_query.resolved.count,
|
|
80
|
-
by_type:
|
|
81
|
-
by_day:
|
|
92
|
+
by_type: volume.by_group_attribute(:error_type).sort_by { |_, count| -count }.to_h,
|
|
93
|
+
by_day: volume.by_day.transform_keys(&:to_s),
|
|
82
94
|
affected_users_incomplete: affected_users_incomplete?
|
|
83
95
|
}
|
|
84
96
|
end
|
|
85
97
|
|
|
86
|
-
#
|
|
87
|
-
#
|
|
98
|
+
# One EventVolume for the window, reused by every EVENT figure below so
|
|
99
|
+
# they cannot drift apart from each other or from the headline total.
|
|
100
|
+
def volume
|
|
101
|
+
@volume ||= Queries::EventVolume.new(base_scope, @start_date)
|
|
102
|
+
end
|
|
103
|
+
|
|
104
|
+
# Total EVENTS in the window. dashboard_stats.rb counts the same way,
|
|
105
|
+
# through the same primitive -- that agreement is asserted by
|
|
106
|
+
# spec/queries/overview_analytics_agreement_spec.rb.
|
|
88
107
|
def event_count
|
|
89
|
-
|
|
108
|
+
volume.count
|
|
90
109
|
end
|
|
91
110
|
|
|
92
111
|
def errors_over_time
|
|
93
|
-
|
|
112
|
+
volume.by_day
|
|
94
113
|
end
|
|
95
114
|
|
|
115
|
+
# Top 10 by EVENTS. Deliberately a partial breakdown -- it does not sum
|
|
116
|
+
# to the headline total, and the agreement spec treats it as such.
|
|
96
117
|
def errors_by_type
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
.to_h
|
|
118
|
+
volume.by_group_attribute(:error_type)
|
|
119
|
+
.sort_by { |_, count| -count }
|
|
120
|
+
.first(10)
|
|
121
|
+
.to_h
|
|
102
122
|
end
|
|
103
123
|
|
|
104
124
|
def errors_by_platform
|
|
105
|
-
|
|
125
|
+
volume.by_group_attribute(:platform)
|
|
106
126
|
end
|
|
107
127
|
|
|
108
128
|
# NULL (captured before the column existed) is reported under :unknown
|
|
@@ -110,14 +130,16 @@ module RailsErrorDashboard
|
|
|
110
130
|
def errors_by_environment
|
|
111
131
|
return {} unless ErrorLog.column_names.include?("environment")
|
|
112
132
|
|
|
113
|
-
|
|
133
|
+
volume.by_group_attribute(:environment).transform_keys { |env| env.nil? ? :unknown : env }
|
|
114
134
|
end
|
|
115
135
|
|
|
136
|
+
# Diurnal pattern: which hour of the day errors peak in, 0..23.
|
|
137
|
+
#
|
|
138
|
+
# Counted per EVENT against the event's own timestamp. Grouping ErrorLog
|
|
139
|
+
# by its first-seen hour put a group's whole lifetime volume on the hour
|
|
140
|
+
# it was first seen, so a recurrence never moved the curve.
|
|
116
141
|
def errors_by_hour
|
|
117
|
-
|
|
118
|
-
# (when in the day errors peak). The chart title says "Errors by Hour
|
|
119
|
-
# of Day" — group_by_hour produced a chronological time series instead.
|
|
120
|
-
base_query.group_by_hour_of_day(:occurred_at).sum(:occurrence_count)
|
|
142
|
+
volume.by_hour_of_day
|
|
121
143
|
end
|
|
122
144
|
|
|
123
145
|
# Events per user, counted from OCCURRENCE rows.
|
|
@@ -149,17 +171,52 @@ module RailsErrorDashboard
|
|
|
149
171
|
|
|
150
172
|
# A user whose events predate occurrence tracking -- or whose events
|
|
151
173
|
# were shed by storm protection, which writes no occurrence row -- still
|
|
152
|
-
# belongs in the table.
|
|
153
|
-
#
|
|
154
|
-
|
|
174
|
+
# belongs in the table. The fallback covers ONLY groups with no
|
|
175
|
+
# occurrence coverage at all in this window.
|
|
176
|
+
#
|
|
177
|
+
# It used to merge by taking the larger of the two counts per user, and
|
|
178
|
+
# that overstated real users: the group's user_id is overwritten by
|
|
179
|
+
# every new occurrence, so group_user_counts attributes a group's whole
|
|
180
|
+
# lifetime count to whoever hit it LAST. For occurrences {A: 2, B: 1}
|
|
181
|
+
# the group figure for B was 3, max kept 3, and the page reported five
|
|
182
|
+
# events for a three-event group. A number that overstates a user's
|
|
183
|
+
# events cannot be presented as an "at least" bound, so the uncovered
|
|
184
|
+
# case is counted and the covered case is left to the occurrence rows.
|
|
185
|
+
uncovered = uncovered_group_user_counts
|
|
186
|
+
counts.merge(uncovered) { |_user, from_occurrences, from_groups| from_occurrences + from_groups }
|
|
155
187
|
rescue StandardError
|
|
156
188
|
group_user_counts
|
|
157
189
|
end
|
|
158
190
|
|
|
191
|
+
# GROUP-based on purpose: this is the fallback for groups with no
|
|
192
|
+
# per-event record, where the group's own user_id and lifetime count are
|
|
193
|
+
# the only evidence that exists. Routing it through EventVolume would be
|
|
194
|
+
# wrong -- EventVolume counts events, and this needs the per-user split
|
|
195
|
+
# that only the group row carries. See design.md D5.
|
|
159
196
|
def group_user_counts
|
|
160
197
|
base_query.where.not(user_id: nil).group(:user_id).sum(:occurrence_count)
|
|
161
198
|
end
|
|
162
199
|
|
|
200
|
+
# Groups in this window that have NO occurrence row at all: rows from
|
|
201
|
+
# before occurrence tracking, and groups whose every event was shed by
|
|
202
|
+
# storm protection. Their occurrence_count is the only evidence those
|
|
203
|
+
# events happened, and the group's own user_id the only attribution
|
|
204
|
+
# available -- a floor, which affected_users_incomplete? reports.
|
|
205
|
+
#
|
|
206
|
+
# A group with even one occurrence row is excluded: its per-event rows
|
|
207
|
+
# are authoritative, and adding the group total on top is what produced
|
|
208
|
+
# the overcount.
|
|
209
|
+
def uncovered_group_user_counts
|
|
210
|
+
covered = ErrorOccurrence.where(error_log_id: base_query.select(:id)).select(:error_log_id)
|
|
211
|
+
|
|
212
|
+
base_query.where.not(user_id: nil)
|
|
213
|
+
.where.not(id: covered)
|
|
214
|
+
.group(:user_id)
|
|
215
|
+
.sum(:occurrence_count)
|
|
216
|
+
rescue StandardError
|
|
217
|
+
{}
|
|
218
|
+
end
|
|
219
|
+
|
|
163
220
|
# Occurrence rows joined back to error_logs, so the application filter
|
|
164
221
|
# (which the occurrence table has no column for) still applies.
|
|
165
222
|
def occurrence_scope
|
|
@@ -214,11 +271,13 @@ module RailsErrorDashboard
|
|
|
214
271
|
end
|
|
215
272
|
|
|
216
273
|
def mobile_errors_count
|
|
217
|
-
|
|
274
|
+
Queries::EventVolume.in_window(base_scope.where(platform: [ "iOS", "Android" ]), @start_date)
|
|
218
275
|
end
|
|
219
276
|
|
|
220
277
|
def api_errors_count
|
|
221
|
-
|
|
278
|
+
Queries::EventVolume.in_window(
|
|
279
|
+
base_scope.where("platform IS NULL OR platform = ?", "API"), @start_date
|
|
280
|
+
)
|
|
222
281
|
end
|
|
223
282
|
|
|
224
283
|
# Pattern insights for top error types
|
|
@@ -23,6 +23,113 @@ module RailsErrorDashboard
|
|
|
23
23
|
new(error_type, platform).weekly_baseline
|
|
24
24
|
end
|
|
25
25
|
|
|
26
|
+
# Pairs considered by .current_anomalies, busiest first. A bound, so the
|
|
27
|
+
# check costs the same whether 5 or 50,000 error types are stored.
|
|
28
|
+
MAX_ANOMALY_PAIRS = 500
|
|
29
|
+
BASELINE_PRECEDENCE = %w[hourly daily weekly].freeze
|
|
30
|
+
|
|
31
|
+
# Every (error_type, platform) pair that is anomalous RIGHT NOW, for all
|
|
32
|
+
# pairs at once. Same rule as #check_current_anomaly -- the first baseline
|
|
33
|
+
# available among hourly / daily / weekly, compared against this hour /
|
|
34
|
+
# today / this week -- but in a fixed number of queries instead of about six
|
|
35
|
+
# per pair. DashboardStats calls this from the live stats broadcast, which
|
|
36
|
+
# runs inside the host app's capture path.
|
|
37
|
+
#
|
|
38
|
+
# Only pairs with an event this week are looked at: a pair with none has a
|
|
39
|
+
# count of zero, which no baseline can call anomalous.
|
|
40
|
+
#
|
|
41
|
+
# @return [Array<Hash>] error_type, platform, count, level, std_devs_above,
|
|
42
|
+
# baseline_type. Empty on any failure -- never raises.
|
|
43
|
+
def self.current_anomalies(sensitivity: 2, application_id: nil)
|
|
44
|
+
return [] unless defined?(ErrorBaseline) && ErrorBaseline.table_exists?
|
|
45
|
+
|
|
46
|
+
counts = current_counts_by_pair(application_id: application_id)
|
|
47
|
+
return [] if counts.empty?
|
|
48
|
+
|
|
49
|
+
baselines = latest_baselines_for(counts.keys)
|
|
50
|
+
|
|
51
|
+
counts.filter_map do |(error_type, platform), windows|
|
|
52
|
+
kind = BASELINE_PRECEDENCE.find { |type| baselines[[ error_type, platform, type ]] }
|
|
53
|
+
next unless kind
|
|
54
|
+
|
|
55
|
+
baseline = baselines[[ error_type, platform, kind ]]
|
|
56
|
+
# A flat history has no spread to measure against; dividing by it
|
|
57
|
+
# makes any count above the mean infinitely anomalous.
|
|
58
|
+
next if baseline.std_dev.nil? || baseline.std_dev.zero?
|
|
59
|
+
|
|
60
|
+
count = windows.fetch(kind.to_sym)
|
|
61
|
+
level = baseline.anomaly_level(count, sensitivity: sensitivity)
|
|
62
|
+
next unless level
|
|
63
|
+
|
|
64
|
+
{
|
|
65
|
+
error_type: error_type,
|
|
66
|
+
platform: platform,
|
|
67
|
+
count: count,
|
|
68
|
+
level: level,
|
|
69
|
+
std_devs_above: baseline.std_devs_above_mean(count),
|
|
70
|
+
baseline_type: kind
|
|
71
|
+
}
|
|
72
|
+
end
|
|
73
|
+
rescue => e
|
|
74
|
+
RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] current_anomalies failed: #{e.class}: #{e.message}")
|
|
75
|
+
[]
|
|
76
|
+
end
|
|
77
|
+
|
|
78
|
+
# { [error_type, platform] => { hourly:, daily:, weekly: } } in ONE grouped
|
|
79
|
+
# query, counted in the units the baselines were built from (occurrence
|
|
80
|
+
# rows when that table exists). The week is the widest window, so it is
|
|
81
|
+
# the WHERE; the day and the hour are conditional sums inside it.
|
|
82
|
+
def self.current_counts_by_pair(application_id: nil)
|
|
83
|
+
logs = ErrorLog.table_name
|
|
84
|
+
column = Services::BaselineCalculator.time_column
|
|
85
|
+
now = Time.current
|
|
86
|
+
|
|
87
|
+
relation = if defined?(ErrorOccurrence) && ErrorOccurrence.table_exists?
|
|
88
|
+
ErrorOccurrence.joins(:error_log)
|
|
89
|
+
else
|
|
90
|
+
ErrorLog.all
|
|
91
|
+
end
|
|
92
|
+
relation = relation.where(logs => { application_id: application_id }) if application_id.present?
|
|
93
|
+
|
|
94
|
+
since = ->(time) { ErrorLog.sanitize_sql_array([ "SUM(CASE WHEN #{column} >= ? THEN 1 ELSE 0 END)", time ]) }
|
|
95
|
+
|
|
96
|
+
rows = relation
|
|
97
|
+
.where("#{column} >= ?", now.beginning_of_week)
|
|
98
|
+
.group("#{logs}.error_type", "#{logs}.platform")
|
|
99
|
+
.order(Arel.sql("COUNT(*) DESC"))
|
|
100
|
+
.limit(MAX_ANOMALY_PAIRS)
|
|
101
|
+
.pluck(Arel.sql("#{logs}.error_type"), Arel.sql("#{logs}.platform"), Arel.sql("COUNT(*)"),
|
|
102
|
+
Arel.sql(since.call(now.beginning_of_day)), Arel.sql(since.call(now.beginning_of_hour)))
|
|
103
|
+
|
|
104
|
+
rows.to_h do |error_type, platform, weekly, daily, hourly|
|
|
105
|
+
[ [ error_type, platform ], { weekly: weekly.to_i, daily: daily.to_i, hourly: hourly.to_i } ]
|
|
106
|
+
end
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
# { [error_type, platform, baseline_type] => ErrorBaseline }, the most recent
|
|
110
|
+
# row of each, in ONE query. Baseline rows accumulate (one per calculation
|
|
111
|
+
# period), so "latest" is resolved in SQL with a join on MAX(period_start)
|
|
112
|
+
# rather than by loading the history. The join form is portable to every
|
|
113
|
+
# adapter; a row-value IN is not.
|
|
114
|
+
def self.latest_baselines_for(pairs)
|
|
115
|
+
table = ErrorBaseline.table_name
|
|
116
|
+
types = pairs.map(&:first).compact.uniq
|
|
117
|
+
return {} if types.empty?
|
|
118
|
+
|
|
119
|
+
latest = ErrorBaseline.where(error_type: types, baseline_type: BASELINE_PRECEDENCE)
|
|
120
|
+
.group(:error_type, :platform, :baseline_type)
|
|
121
|
+
.select(:error_type, :platform, :baseline_type, "MAX(period_start) AS latest_period_start")
|
|
122
|
+
|
|
123
|
+
ErrorBaseline
|
|
124
|
+
.joins("INNER JOIN (#{latest.to_sql}) latest_baselines ON " \
|
|
125
|
+
"latest_baselines.error_type = #{table}.error_type AND " \
|
|
126
|
+
"latest_baselines.platform = #{table}.platform AND " \
|
|
127
|
+
"latest_baselines.baseline_type = #{table}.baseline_type AND " \
|
|
128
|
+
"latest_baselines.latest_period_start = #{table}.period_start")
|
|
129
|
+
.index_by { |baseline| [ baseline.error_type, baseline.platform, baseline.baseline_type ] }
|
|
130
|
+
end
|
|
131
|
+
private_class_method :current_counts_by_pair, :latest_baselines_for
|
|
132
|
+
|
|
26
133
|
def initialize(error_type, platform)
|
|
27
134
|
@error_type = error_type
|
|
28
135
|
@platform = platform
|