rails_error_dashboard 0.12.1 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. checksums.yaml +4 -4
  2. data/app/controllers/rails_error_dashboard/application_controller.rb +92 -11
  3. data/app/controllers/rails_error_dashboard/errors_controller.rb +111 -34
  4. data/app/controllers/rails_error_dashboard/webhooks_controller.rb +3 -0
  5. data/app/helpers/rails_error_dashboard/application_helper.rb +46 -1
  6. data/app/helpers/rails_error_dashboard/backtrace_helper.rb +8 -0
  7. data/app/jobs/rails_error_dashboard/async_error_logging_job.rb +10 -0
  8. data/app/jobs/rails_error_dashboard/concerns/plain_channel_message.rb +71 -0
  9. data/app/jobs/rails_error_dashboard/notification_burst_summary_job.rb +55 -0
  10. data/app/jobs/rails_error_dashboard/retention_cleanup_job.rb +122 -1
  11. data/app/jobs/rails_error_dashboard/storm_flush_job.rb +7 -4
  12. data/app/jobs/rails_error_dashboard/storm_notification_job.rb +8 -44
  13. data/app/models/rails_error_dashboard/error_baseline.rb +12 -9
  14. data/app/models/rails_error_dashboard/error_comment.rb +0 -5
  15. data/app/models/rails_error_dashboard/error_log.rb +19 -3
  16. data/app/models/rails_error_dashboard/error_logs_record.rb +34 -0
  17. data/app/models/rails_error_dashboard/error_occurrence.rb +9 -1
  18. data/app/models/rails_error_dashboard/event_count.rb +132 -0
  19. data/app/models/rails_error_dashboard/event_timing_gap.rb +55 -0
  20. data/app/views/layouts/rails_error_dashboard.html.erb +58 -5
  21. data/app/views/rails_error_dashboard/errors/_discussion.html.erb +18 -11
  22. data/app/views/rails_error_dashboard/errors/_issue_section.html.erb +22 -8
  23. data/app/views/rails_error_dashboard/errors/_request_context.html.erb +2 -0
  24. data/app/views/rails_error_dashboard/errors/_sidebar_metadata.html.erb +15 -6
  25. data/app/views/rails_error_dashboard/errors/analytics.html.erb +8 -8
  26. data/app/views/rails_error_dashboard/errors/diagnostic_dumps.html.erb +1 -1
  27. data/app/views/rails_error_dashboard/errors/index.html.erb +6 -1
  28. data/app/views/rails_error_dashboard/errors/overview.html.erb +12 -0
  29. data/app/views/rails_error_dashboard/errors/platform_comparison.html.erb +7 -7
  30. data/app/views/rails_error_dashboard/errors/releases.html.erb +2 -2
  31. data/app/views/rails_error_dashboard/errors/settings.html.erb +2 -0
  32. data/app/views/rails_error_dashboard/errors/show.html.erb +3 -3
  33. data/config/locales/de.yml +30 -0
  34. data/config/locales/en.yml +37 -0
  35. data/config/locales/es.yml +30 -0
  36. data/config/locales/fr.yml +30 -0
  37. data/config/locales/it.yml +30 -0
  38. data/config/locales/ja.yml +30 -0
  39. data/config/locales/pl.yml +30 -0
  40. data/config/locales/pt-BR.yml +30 -0
  41. data/config/locales/ru.yml +30 -0
  42. data/config/locales/uk.yml +30 -0
  43. data/config/locales/zh-CN.yml +30 -0
  44. data/db/migrate/20260917000001_add_last_notified_at_to_error_logs.rb +40 -0
  45. data/db/migrate/20260919000001_create_event_counts.rb +71 -0
  46. data/db/migrate/20260920000001_add_buckets_incomplete_to_storm_events.rb +25 -0
  47. data/db/migrate/20260920000002_create_event_timing_gaps.rb +55 -0
  48. data/lib/generators/rails_error_dashboard/install/templates/initializer.rb +12 -1
  49. data/lib/rails_error_dashboard/commands/assign_error.rb +12 -2
  50. data/lib/rails_error_dashboard/commands/backfill_environments.rb +2 -0
  51. data/lib/rails_error_dashboard/commands/backfill_resolved_at.rb +42 -0
  52. data/lib/rails_error_dashboard/commands/batch_delete_errors.rb +1 -0
  53. data/lib/rails_error_dashboard/commands/batch_mute_errors.rb +2 -0
  54. data/lib/rails_error_dashboard/commands/batch_resolve_errors.rb +2 -0
  55. data/lib/rails_error_dashboard/commands/batch_unmute_errors.rb +2 -0
  56. data/lib/rails_error_dashboard/commands/find_or_increment_error.rb +109 -11
  57. data/lib/rails_error_dashboard/commands/flush_rack_attack_events.rb +3 -1
  58. data/lib/rails_error_dashboard/commands/flush_storm_counts.rb +252 -12
  59. data/lib/rails_error_dashboard/commands/flush_swallowed_exceptions.rb +5 -2
  60. data/lib/rails_error_dashboard/commands/link_existing_issue.rb +1 -0
  61. data/lib/rails_error_dashboard/commands/log_error.rb +291 -40
  62. data/lib/rails_error_dashboard/commands/mute_error.rb +1 -0
  63. data/lib/rails_error_dashboard/commands/resolve_error.rb +2 -0
  64. data/lib/rails_error_dashboard/commands/scrub_invalid_encoding.rb +104 -0
  65. data/lib/rails_error_dashboard/commands/snooze_error.rb +34 -10
  66. data/lib/rails_error_dashboard/commands/unmute_error.rb +1 -0
  67. data/lib/rails_error_dashboard/commands/update_error_priority.rb +23 -2
  68. data/lib/rails_error_dashboard/commands/update_error_status.rb +33 -6
  69. data/lib/rails_error_dashboard/configuration.rb +39 -1
  70. data/lib/rails_error_dashboard/engine.rb +28 -0
  71. data/lib/rails_error_dashboard/manual_error_reporter.rb +16 -5
  72. data/lib/rails_error_dashboard/queries/analytics_stats.rb +89 -30
  73. data/lib/rails_error_dashboard/queries/baseline_stats.rb +107 -0
  74. data/lib/rails_error_dashboard/queries/dashboard_stats.rb +216 -74
  75. data/lib/rails_error_dashboard/queries/error_correlation.rb +6 -3
  76. data/lib/rails_error_dashboard/queries/errors_list.rb +15 -2
  77. data/lib/rails_error_dashboard/queries/event_volume.rb +503 -0
  78. data/lib/rails_error_dashboard/queries/similar_errors.rb +1 -1
  79. data/lib/rails_error_dashboard/services/analytics_cache_manager.rb +43 -18
  80. data/lib/rails_error_dashboard/services/backtrace_processor.rb +3 -1
  81. data/lib/rails_error_dashboard/services/breadcrumb_collector.rb +23 -0
  82. data/lib/rails_error_dashboard/services/cause_chain_extractor.rb +3 -1
  83. data/lib/rails_error_dashboard/services/codeberg_issue_client.rb +13 -2
  84. data/lib/rails_error_dashboard/services/diagnostic_dump_generator.rb +5 -3
  85. data/lib/rails_error_dashboard/services/encoding_sanitizer.rb +80 -0
  86. data/lib/rails_error_dashboard/services/error_broadcaster.rb +180 -32
  87. data/lib/rails_error_dashboard/services/error_hash_generator.rb +9 -3
  88. data/lib/rails_error_dashboard/services/error_notification_dispatcher.rb +13 -0
  89. data/lib/rails_error_dashboard/services/exception_filter.rb +55 -0
  90. data/lib/rails_error_dashboard/services/git_head_reader.rb +102 -0
  91. data/lib/rails_error_dashboard/services/notification_throttler.rb +172 -28
  92. data/lib/rails_error_dashboard/services/sensitive_data_filter.rb +47 -1
  93. data/lib/rails_error_dashboard/services/storm_protection/circuit_breaker.rb +59 -3
  94. data/lib/rails_error_dashboard/services/storm_protection/count_buffer.rb +64 -6
  95. data/lib/rails_error_dashboard/services/storm_protection/fingerprint_buckets.rb +25 -2
  96. data/lib/rails_error_dashboard/services/storm_protection/gate.rb +32 -5
  97. data/lib/rails_error_dashboard/services/swallowed_exception_tracker.rb +99 -23
  98. data/lib/rails_error_dashboard/services/url_safety.rb +40 -0
  99. data/lib/rails_error_dashboard/services/variable_serializer.rb +125 -10
  100. data/lib/rails_error_dashboard/subscribers/breadcrumb_subscriber.rb +111 -0
  101. data/lib/rails_error_dashboard/subscribers/issue_tracker_subscriber.rb +11 -2
  102. data/lib/rails_error_dashboard/value_objects/error_context.rb +40 -3
  103. data/lib/rails_error_dashboard/version.rb +1 -1
  104. data/lib/rails_error_dashboard.rb +34 -0
  105. data/lib/tasks/error_dashboard.rake +54 -4
  106. metadata +16 -2
@@ -4,7 +4,14 @@ module RailsErrorDashboard
4
4
  module Commands
5
5
  # Command: Snooze an error for a given number of hours
6
6
  # This is a write operation that sets snoozed_until and optionally creates a comment
7
+ # Returns {success: bool, error: ErrorLog}; a failure also carries
8
+ # reason: :invalid_hours and writes nothing.
7
9
  class SnoozeError
10
+ # 30 days. The form offers at most a week; the cap is for anything that
11
+ # does not come from the form. Without one, a negative value "snoozed"
12
+ # into the past and a huge one overflowed the timestamp.
13
+ MAX_SNOOZE_HOURS = 720
14
+
8
15
  def self.call(error_id, hours:, reason: nil)
9
16
  new(error_id, hours, reason).call
10
17
  end
@@ -17,18 +24,35 @@ module RailsErrorDashboard
17
24
 
18
25
  def call
19
26
  error = ErrorLog.find(@error_id)
20
- snooze_until = @hours.hours.from_now
21
-
22
- # Store snooze reason in comments if provided
23
- if @reason.present?
24
- error.comments.create!(
25
- author_name: error.assigned_to || "System",
26
- body: "Snoozed for #{@hours} hours: #{@reason}"
27
- )
27
+ hours = whole_hours(@hours)
28
+
29
+ unless hours && (1..MAX_SNOOZE_HOURS).cover?(hours)
30
+ return { success: false, error: error, reason: :invalid_hours }
31
+ end
32
+
33
+ error.transaction do
34
+ if @reason.present?
35
+ error.comments.create!(
36
+ author_name: error.assigned_to || "System",
37
+ body: "Snoozed for #{hours} hours: #{@reason}"
38
+ )
39
+ end
40
+
41
+ error.update!(snoozed_until: hours.hours.from_now)
28
42
  end
29
43
 
30
- error.update!(snoozed_until: snooze_until)
31
- error
44
+ { success: true, error: error }
45
+ end
46
+
47
+ private
48
+
49
+ # An Integer, or a String that is one ("24"). Anything else -- a Float, a
50
+ # nested parameter, "abc" -- is nil rather than a guess.
51
+ def whole_hours(value)
52
+ case value
53
+ when Integer then value
54
+ when String then Integer(value.strip, 10, exception: false)
55
+ end
32
56
  end
33
57
  end
34
58
  end
@@ -22,6 +22,7 @@ module RailsErrorDashboard
22
22
  muted_reason: nil
23
23
  )
24
24
 
25
+ Services::AnalyticsCacheManager.clear
25
26
  PluginRegistry.dispatch(:on_error_unmuted, error)
26
27
  error
27
28
  end
@@ -4,6 +4,8 @@ module RailsErrorDashboard
4
4
  module Commands
5
5
  # Command: Update the priority level of an error
6
6
  # This is a write operation that updates the priority_level field on an ErrorLog record
7
+ # Returns {success: bool, error: ErrorLog}; a failure also carries
8
+ # reason: :invalid_priority and leaves the existing priority alone.
7
9
  class UpdateErrorPriority
8
10
  def self.call(error_id, priority_level:)
9
11
  new(error_id, priority_level).call
@@ -16,8 +18,27 @@ module RailsErrorDashboard
16
18
 
17
19
  def call
18
20
  error = ErrorLog.find(@error_id)
19
- error.update!(priority_level: @priority_level)
20
- error
21
+ level = whole_number(@priority_level)
22
+
23
+ # The column is an integer, so an unchecked "x" was cast and stored as
24
+ # 0 -- silently replacing a real priority with Low.
25
+ unless ErrorLog::PRIORITY_LEVELS.key?(level)
26
+ return { success: false, error: error, reason: :invalid_priority }
27
+ end
28
+
29
+ error.update!(priority_level: level)
30
+ { success: true, error: error }
31
+ end
32
+
33
+ private
34
+
35
+ # An Integer, or a String that is one ("3"). A nested parameter, a Float
36
+ # or free text is nil.
37
+ def whole_number(value)
38
+ case value
39
+ when Integer then value
40
+ when String then Integer(value.strip, 10, exception: false)
41
+ end
21
42
  end
22
43
  end
23
44
  end
@@ -4,7 +4,8 @@ module RailsErrorDashboard
4
4
  module Commands
5
5
  # Command: Update the status of an error with optional comment
6
6
  # This is a write operation that validates transitions and updates status
7
- # Returns {success: bool, error: ErrorLog}
7
+ # Returns {success: bool, error: ErrorLog}; a failure also carries
8
+ # reason: :unknown_status or :invalid_transition so the caller can say which.
8
9
  class UpdateErrorStatus
9
10
  def self.call(error_id, status:, comment: nil)
10
11
  new(error_id, status, comment).call
@@ -19,15 +20,17 @@ module RailsErrorDashboard
19
20
  def call
20
21
  error = ErrorLog.find(@error_id)
21
22
 
23
+ # A nested param arrives as a Hash-like object, never a known status.
24
+ unless @status.is_a?(String) && ErrorLog::STATUSES.include?(@status)
25
+ return { success: false, error: error, reason: :unknown_status }
26
+ end
27
+
22
28
  unless error.can_transition_to?(@status)
23
- return { success: false, error: error }
29
+ return { success: false, error: error, reason: :invalid_transition }
24
30
  end
25
31
 
26
32
  error.transaction do
27
- error.update!(status: @status)
28
-
29
- # Auto-resolve if status is "resolved"
30
- error.update!(resolved: true) if @status == "resolved"
33
+ error.update!(status_attributes(error))
31
34
 
32
35
  # Add comment about status change
33
36
  if @comment.present?
@@ -38,8 +41,32 @@ module RailsErrorDashboard
38
41
  end
39
42
  end
40
43
 
44
+ # The stat cards are cached; a user action must show up at once.
45
+ Services::AnalyticsCacheManager.clear
46
+
41
47
  { success: true, error: error }
42
48
  end
49
+
50
+ private
51
+
52
+ # One write, so the three columns can never disagree. resolved_at is what
53
+ # MTTR is computed from: leaving it nil (as this command used to) dropped
54
+ # every error resolved through the status workflow from the MTTR figures,
55
+ # and leaving it set on a reopened error kept a stale resolution time.
56
+ # Only "resolved" sets the flag -- wont_fix stays resolved: false.
57
+ def status_attributes(error)
58
+ attrs = { status: @status }
59
+
60
+ if @status == "resolved"
61
+ attrs[:resolved] = true
62
+ attrs[:resolved_at] = Time.current
63
+ elsif error.status == "resolved" || error.resolved?
64
+ attrs[:resolved] = false
65
+ attrs[:resolved_at] = nil
66
+ end
67
+
68
+ attrs
69
+ end
43
70
  end
44
71
  end
45
72
  end
@@ -157,6 +157,8 @@ module RailsErrorDashboard
157
157
  attr_accessor :notification_minimum_severity # Minimum severity to notify (default: :low = notify all)
158
158
  attr_accessor :notification_cooldown_minutes # Per-error cooldown in minutes (default: 5, 0 = disabled)
159
159
  attr_accessor :notification_threshold_alerts # Occurrence milestones that trigger notification (default: [10, 50, 100, 500, 1000])
160
+ attr_accessor :notification_burst_limit # Max FIRST-OCCURRENCE notifications per window, per process (default: 10, 0 = no cap)
161
+ attr_accessor :notification_burst_window_seconds # Length of that window in seconds (default: 60)
160
162
 
161
163
  # Breadcrumbs (request activity trail)
162
164
  attr_accessor :enable_breadcrumbs # Master switch (default: false)
@@ -180,6 +182,12 @@ module RailsErrorDashboard
180
182
  attr_accessor :local_variable_max_array_items # Max array items to serialize (default: 10)
181
183
  attr_accessor :local_variable_max_hash_items # Max hash entries to serialize (default: 20)
182
184
  attr_accessor :local_variable_filter_patterns # Additional sensitive name patterns (default: [])
185
+ # Calling #inspect on an unknown object runs arbitrary APPLICATION code on
186
+ # the failure path; truncating the result bounds storage, not cost. The
187
+ # default is a safe structural summary, with full inspect per type by
188
+ # opt-in and a wall-clock budget even then.
189
+ attr_accessor :local_variable_inspect_allowlist # Class names whose #inspect may run (default: safe built-ins)
190
+ attr_accessor :local_variable_inspect_budget_ms # Wall-clock budget for one #inspect (default: 5)
183
191
 
184
192
  # Instance variable capture from tp.self (receiver object at raise time)
185
193
  attr_accessor :enable_instance_variables # Master switch (default: false)
@@ -302,7 +310,8 @@ module RailsErrorDashboard
302
310
 
303
311
  @use_separate_database = ENV.fetch("USE_SEPARATE_ERROR_DB", "false") == "true"
304
312
 
305
- # Retention policy - days to keep errors before automatic deletion (default: 90)
313
+ # Retention policy - days an error may go unseen (last_seen_at) before it is
314
+ # deleted automatically (default: 90). An error still occurring is kept.
306
315
  # Set to nil to keep errors forever (not recommended for production)
307
316
  # Schedule cleanup: RailsErrorDashboard::RetentionCleanupJob.perform_later
308
317
  @retention_days = 90
@@ -384,6 +393,8 @@ module RailsErrorDashboard
384
393
  @notification_minimum_severity = :low # Notify on all severities (current behavior)
385
394
  @notification_cooldown_minutes = 5 # 5 min cooldown per error_hash (0 = disabled)
386
395
  @notification_threshold_alerts = [ 10, 50, 100, 500, 1000 ] # Occurrence milestones
396
+ @notification_burst_limit = 10 # New-error notifications per window, per process (0 = no cap)
397
+ @notification_burst_window_seconds = 60 # One summary message replaces the rest of the window
387
398
 
388
399
  # Breadcrumbs defaults - OFF by default (opt-in)
389
400
  @enable_breadcrumbs = false # Master switch
@@ -410,6 +421,20 @@ module RailsErrorDashboard
410
421
  @local_variable_max_array_items = 10 # Max array items to serialize
411
422
  @local_variable_max_hash_items = 20 # Max hash entries to serialize
412
423
  @local_variable_filter_patterns = [] # Additional sensitive variable name patterns
424
+ # Struct is serialized MEMBER-WISE (never via its own #inspect), so it
425
+ # needs no allowlist entry -- see VariableSerializer.serialize_struct.
426
+ #
427
+ # ActiveModel is deliberately NOT allowlisted. Reading
428
+ # ActiveModel::Attributes#attributes runs each attribute's type cast,
429
+ # which is application code, so neither #inspect nor member-wise reading
430
+ # can be bounded for it; it gets a safe summary instead.
431
+ #
432
+ # Anything added here opts that type IN to unbounded execution: the only
433
+ # way to interrupt arbitrary Ruby mid-call is Timeout, which is not safe
434
+ # on the capture path. The budget below selects the stored OUTPUT after
435
+ # the fact; it does not bound the work.
436
+ @local_variable_inspect_allowlist = []
437
+ @local_variable_inspect_budget_ms = 5
413
438
 
414
439
  # Instance variable capture defaults - OFF by default (opt-in)
415
440
  @enable_instance_variables = false # Capture ivars from tp.self at raise time
@@ -854,6 +879,19 @@ module RailsErrorDashboard
854
879
  errors << "notification_threshold_alerts must be an Array (got: #{notification_threshold_alerts.class})"
855
880
  end
856
881
 
882
+ # Validate the first-occurrence burst cap (non-negative integers; 0 or nil
883
+ # turns the cap off)
884
+ {
885
+ notification_burst_limit: notification_burst_limit,
886
+ notification_burst_window_seconds: notification_burst_window_seconds
887
+ }.each do |name, value|
888
+ next if value.nil?
889
+
890
+ unless value.is_a?(Integer) && value >= 0
891
+ errors << "#{name} must be a non-negative Integer (got: #{value.inspect})"
892
+ end
893
+ end
894
+
857
895
  # Log warnings (non-fatal issues)
858
896
  warnings.each do |warning|
859
897
  Rails.logger.warn "[Rails Error Dashboard] #{warning}" if defined?(Rails) && Rails.respond_to?(:logger) && Rails.logger
@@ -71,6 +71,11 @@ module RailsErrorDashboard
71
71
  next
72
72
  end
73
73
 
74
+ # Resolve the running commit now (three small file reads, memoised, never
75
+ # raises) so that no capture ever pays for it. Skipped when the SHA is
76
+ # configured: then it is never consulted.
77
+ RailsErrorDashboard.detected_git_sha if RailsErrorDashboard.configuration.git_sha.blank?
78
+
74
79
  if RailsErrorDashboard.configuration.enable_error_subscriber
75
80
  Rails.error.subscribe(RailsErrorDashboard::ErrorReporter.new)
76
81
  end
@@ -80,6 +85,19 @@ module RailsErrorDashboard
80
85
  RailsErrorDashboard::Subscribers::BreadcrumbSubscriber.subscribe!
81
86
  end
82
87
 
88
+ # Give background jobs a breadcrumb buffer of their own. init_buffer had
89
+ # exactly one caller -- the Rack middleware -- so a job never entered the
90
+ # HTTP stack and had no buffer at all; every subscriber early-returns on
91
+ # `unless current_buffer`, so a failing job's SQL and custom crumbs were
92
+ # dropped in the one place an error is hardest to reproduce.
93
+ #
94
+ # Registered unconditionally and gated at PERFORM time, not here:
95
+ # enable_breadcrumbs defaults to false, and the ActiveJob callback list
96
+ # is fixed once the class loads, so a boot-time gate would leave this
97
+ # permanently unregistered for any host that turns the feature on in an
98
+ # initializer that runs later. Off, it costs one config read per job.
99
+ RailsErrorDashboard::Subscribers::BreadcrumbSubscriber.install_job_buffer!
100
+
83
101
  # Subscribe to Rack Attack AS::Notifications events (requires Rack::Attack).
84
102
  # Breadcrumbs are NOT required — events persist to their own table (issue #143).
85
103
  if RailsErrorDashboard.configuration.enable_rack_attack_tracking &&
@@ -188,6 +206,16 @@ module RailsErrorDashboard
188
206
  # Enable TracePoint(:raise) + TracePoint(:rescue) for swallowed exception detection
189
207
  if RailsErrorDashboard.configuration.detect_swallowed_exceptions
190
208
  RailsErrorDashboard::Services::SwallowedExceptionTracker.enable!
209
+
210
+ # Drain buffered counts at the end of every request and job. Without
211
+ # this the buffer is only ever drained by a LATER rescue on the SAME
212
+ # thread, so a swallowed exception that happens once stays invisible
213
+ # until the process exits. to_complete fires after the response body
214
+ # is closed, so it never delays a request (safety rule 2); the flush is
215
+ # deadline-gated, so a flood is still one write per interval.
216
+ Rails.application.executor.to_complete do
217
+ RailsErrorDashboard::Services::SwallowedExceptionTracker.flush_if_due!
218
+ end
191
219
  end
192
220
 
193
221
  # Import crash files from previous process death, then register at_exit hook
@@ -42,7 +42,11 @@ module RailsErrorDashboard
42
42
  # @param app_version [String, nil] Version of the app where error occurred
43
43
  # @param metadata [Hash, nil] Additional custom metadata about the error
44
44
  # @param occurred_at [Time, nil] When the error occurred (defaults to Time.current)
45
- # @param severity [Symbol, nil] Severity level (:critical, :high, :medium, :low)
45
+ # @param severity [Symbol, nil] IGNORED, and kept only so existing callers
46
+ # do not break. Severity is not stored: ErrorLog#severity is CLASSIFIED
47
+ # from error_type by Services::SeverityClassifier, so the way to control
48
+ # it is the error_type you report. Accepting this silently made callers
49
+ # believe they had set a value that was never read.
46
50
  # @param source [String, nil] Source identifier (e.g., "frontend", "mobile_app")
47
51
  #
48
52
  # @return [ErrorLog, nil] The created error log record, or nil if filtered/ignored
@@ -65,8 +69,7 @@ module RailsErrorDashboard
65
69
  # user_agent: request.user_agent,
66
70
  # ip_address: request.remote_ip,
67
71
  # app_version: "1.2.3",
68
- # metadata: { card_type: "visa", amount: 99.99 },
69
- # severity: :high
72
+ # metadata: { card_type: "visa", amount: 99.99 }
70
73
  # )
71
74
  def self.report(
72
75
  error_type:,
@@ -100,10 +103,18 @@ module RailsErrorDashboard
100
103
  platform: platform,
101
104
  app_version: app_version,
102
105
  metadata: metadata,
103
- occurred_at: occurred_at || Time.current,
104
- severity: severity
106
+ occurred_at: occurred_at || Time.current
105
107
  }.compact # Remove nil values
106
108
 
109
+ # severity is deliberately NOT forwarded: nothing downstream reads it,
110
+ # and passing it on would keep the illusion that it does.
111
+ if severity.present?
112
+ RailsErrorDashboard::Logger.debug(
113
+ "[RailsErrorDashboard] ManualErrorReporter: severity: is ignored — " \
114
+ "severity is classified from error_type (#{error_type})."
115
+ )
116
+ end
117
+
107
118
  # Use the existing LogError command
108
119
  Commands::LogError.call(synthetic_exception, context)
109
120
  end
@@ -17,7 +17,7 @@ module RailsErrorDashboard
17
17
 
18
18
  def call
19
19
  # Cache analytics data for 5 minutes to reduce database load
20
- # Cache key includes days parameter and last error update timestamp
20
+ # Cache key includes the days parameter and the cache generation
21
21
  Rails.cache.fetch(cache_key, expires_in: 5.minutes) do
22
22
  {
23
23
  days: @days,
@@ -41,13 +41,14 @@ module RailsErrorDashboard
41
41
  # - Query class name
42
42
  # - Days parameter (different time ranges = different caches)
43
43
  # - Application ID (per-app caching)
44
- # - Last error update timestamp (auto-invalidates when errors change)
44
+ # - The cache generation (bumped by user actions; see AnalyticsCacheManager).
45
+ # Captures do not bump it: they rely on the 5-minute TTL.
45
46
  # - Start date (ensures correct time window)
46
47
  [
47
48
  "analytics_stats",
48
49
  @days,
49
50
  @application_id || "all",
50
- base_scope.maximum(:updated_at)&.to_i || 0,
51
+ Services::AnalyticsCacheManager.generation,
51
52
  @start_date.to_date.to_s
52
53
  ].join("/")
53
54
  end
@@ -66,43 +67,62 @@ module RailsErrorDashboard
66
67
 
67
68
  # Two counting units, kept distinct on purpose.
68
69
  #
69
- # An ErrorLog row is a GROUP; its occurrence_count says how many times
70
- # that error actually happened. Volume figures are EVENTS (the sum), so
71
- # this page agrees with the Overview, which counts the same way. Resolved
72
- # and unresolved stay GROUP counts -- a group is the thing that gets
73
- # resolved, and an event cannot be.
70
+ # An ErrorLog row is a GROUP; its occurred_at is FIRST-SEEN and is never
71
+ # rewritten when the error recurs.
72
+ #
73
+ # EVENT figures -> Queries::EventVolume, which counts occurrence rows
74
+ # + storm buckets + the untracked remainder, each
75
+ # against its OWN timestamp.
76
+ # GROUP figures -> base_query, which selects groups by first-seen.
77
+ # Correct here: a group is the thing that gets
78
+ # resolved, and an event cannot be.
79
+ #
80
+ # Mixing them is what made this page disagree with the Overview: filtering
81
+ # GROUPS by first-seen and then summing their LIFETIME occurrence_count
82
+ # answers "how much total volume do the groups born in this window carry",
83
+ # not "how many events happened in this window". A group first seen in
84
+ # August that recurred today was excluded entirely, so the page reported
85
+ # zero events while its own affected-users table listed the very event.
74
86
  def error_statistics
75
87
  {
76
88
  total: event_count,
77
89
  total_groups: base_query.count,
78
90
  unresolved: base_query.unresolved.count,
79
91
  resolved: base_query.resolved.count,
80
- by_type: base_query.group(:error_type).sum(:occurrence_count).sort_by { |_, count| -count }.to_h,
81
- by_day: base_query.group("DATE(occurred_at)").sum(:occurrence_count),
92
+ by_type: volume.by_group_attribute(:error_type).sort_by { |_, count| -count }.to_h,
93
+ by_day: volume.by_day.transform_keys(&:to_s),
82
94
  affected_users_incomplete: affected_users_incomplete?
83
95
  }
84
96
  end
85
97
 
86
- # Total EVENTS in the window. dashboard_stats.rb and
87
- # platform_comparison.rb:165 count the same way.
98
+ # One EventVolume for the window, reused by every EVENT figure below so
99
+ # they cannot drift apart from each other or from the headline total.
100
+ def volume
101
+ @volume ||= Queries::EventVolume.new(base_scope, @start_date)
102
+ end
103
+
104
+ # Total EVENTS in the window. dashboard_stats.rb counts the same way,
105
+ # through the same primitive -- that agreement is asserted by
106
+ # spec/queries/overview_analytics_agreement_spec.rb.
88
107
  def event_count
89
- base_query.sum(:occurrence_count)
108
+ volume.count
90
109
  end
91
110
 
92
111
  def errors_over_time
93
- base_query.group_by_day(:occurred_at).sum(:occurrence_count)
112
+ volume.by_day
94
113
  end
95
114
 
115
+ # Top 10 by EVENTS. Deliberately a partial breakdown -- it does not sum
116
+ # to the headline total, and the agreement spec treats it as such.
96
117
  def errors_by_type
97
- base_query.group(:error_type)
98
- .sum(:occurrence_count)
99
- .sort_by { |_, count| -count }
100
- .first(10)
101
- .to_h
118
+ volume.by_group_attribute(:error_type)
119
+ .sort_by { |_, count| -count }
120
+ .first(10)
121
+ .to_h
102
122
  end
103
123
 
104
124
  def errors_by_platform
105
- base_query.group(:platform).sum(:occurrence_count)
125
+ volume.by_group_attribute(:platform)
106
126
  end
107
127
 
108
128
  # NULL (captured before the column existed) is reported under :unknown
@@ -110,14 +130,16 @@ module RailsErrorDashboard
110
130
  def errors_by_environment
111
131
  return {} unless ErrorLog.column_names.include?("environment")
112
132
 
113
- base_query.group(:environment).sum(:occurrence_count).transform_keys { |env| env.nil? ? :unknown : env }
133
+ volume.by_group_attribute(:environment).transform_keys { |env| env.nil? ? :unknown : env }
114
134
  end
115
135
 
136
+ # Diurnal pattern: which hour of the day errors peak in, 0..23.
137
+ #
138
+ # Counted per EVENT against the event's own timestamp. Grouping ErrorLog
139
+ # by its first-seen hour put a group's whole lifetime volume on the hour
140
+ # it was first seen, so a recurrence never moved the curve.
116
141
  def errors_by_hour
117
- # group_by_hour_of_day buckets into 0..23 to show diurnal patterns
118
- # (when in the day errors peak). The chart title says "Errors by Hour
119
- # of Day" — group_by_hour produced a chronological time series instead.
120
- base_query.group_by_hour_of_day(:occurred_at).sum(:occurrence_count)
142
+ volume.by_hour_of_day
121
143
  end
122
144
 
123
145
  # Events per user, counted from OCCURRENCE rows.
@@ -149,17 +171,52 @@ module RailsErrorDashboard
149
171
 
150
172
  # A user whose events predate occurrence tracking -- or whose events
151
173
  # were shed by storm protection, which writes no occurrence row -- still
152
- # belongs in the table. Fall back to the group's own user for those,
153
- # taking whichever count is larger. UserImpactSummary merges the same way.
154
- group_user_counts.merge(counts) { |_user, group_count, occurrence_count| [ group_count, occurrence_count ].max }
174
+ # belongs in the table. The fallback covers ONLY groups with no
175
+ # occurrence coverage at all in this window.
176
+ #
177
+ # It used to merge by taking the larger of the two counts per user, and
178
+ # that overstated real users: the group's user_id is overwritten by
179
+ # every new occurrence, so group_user_counts attributes a group's whole
180
+ # lifetime count to whoever hit it LAST. For occurrences {A: 2, B: 1}
181
+ # the group figure for B was 3, max kept 3, and the page reported five
182
+ # events for a three-event group. A number that overstates a user's
183
+ # events cannot be presented as an "at least" bound, so the uncovered
184
+ # case is counted and the covered case is left to the occurrence rows.
185
+ uncovered = uncovered_group_user_counts
186
+ counts.merge(uncovered) { |_user, from_occurrences, from_groups| from_occurrences + from_groups }
155
187
  rescue StandardError
156
188
  group_user_counts
157
189
  end
158
190
 
191
+ # GROUP-based on purpose: this is the fallback for groups with no
192
+ # per-event record, where the group's own user_id and lifetime count are
193
+ # the only evidence that exists. Routing it through EventVolume would be
194
+ # wrong -- EventVolume counts events, and this needs the per-user split
195
+ # that only the group row carries. See design.md D5.
159
196
  def group_user_counts
160
197
  base_query.where.not(user_id: nil).group(:user_id).sum(:occurrence_count)
161
198
  end
162
199
 
200
+ # Groups in this window that have NO occurrence row at all: rows from
201
+ # before occurrence tracking, and groups whose every event was shed by
202
+ # storm protection. Their occurrence_count is the only evidence those
203
+ # events happened, and the group's own user_id the only attribution
204
+ # available -- a floor, which affected_users_incomplete? reports.
205
+ #
206
+ # A group with even one occurrence row is excluded: its per-event rows
207
+ # are authoritative, and adding the group total on top is what produced
208
+ # the overcount.
209
+ def uncovered_group_user_counts
210
+ covered = ErrorOccurrence.where(error_log_id: base_query.select(:id)).select(:error_log_id)
211
+
212
+ base_query.where.not(user_id: nil)
213
+ .where.not(id: covered)
214
+ .group(:user_id)
215
+ .sum(:occurrence_count)
216
+ rescue StandardError
217
+ {}
218
+ end
219
+
163
220
  # Occurrence rows joined back to error_logs, so the application filter
164
221
  # (which the occurrence table has no column for) still applies.
165
222
  def occurrence_scope
@@ -214,11 +271,13 @@ module RailsErrorDashboard
214
271
  end
215
272
 
216
273
  def mobile_errors_count
217
- base_query.where(platform: [ "iOS", "Android" ]).sum(:occurrence_count)
274
+ Queries::EventVolume.in_window(base_scope.where(platform: [ "iOS", "Android" ]), @start_date)
218
275
  end
219
276
 
220
277
  def api_errors_count
221
- base_query.where("platform IS NULL OR platform = ?", "API").sum(:occurrence_count)
278
+ Queries::EventVolume.in_window(
279
+ base_scope.where("platform IS NULL OR platform = ?", "API"), @start_date
280
+ )
222
281
  end
223
282
 
224
283
  # Pattern insights for top error types
@@ -23,6 +23,113 @@ module RailsErrorDashboard
23
23
  new(error_type, platform).weekly_baseline
24
24
  end
25
25
 
26
+ # Pairs considered by .current_anomalies, busiest first. A bound, so the
27
+ # check costs the same whether 5 or 50,000 error types are stored.
28
+ MAX_ANOMALY_PAIRS = 500
29
+ BASELINE_PRECEDENCE = %w[hourly daily weekly].freeze
30
+
31
+ # Every (error_type, platform) pair that is anomalous RIGHT NOW, for all
32
+ # pairs at once. Same rule as #check_current_anomaly -- the first baseline
33
+ # available among hourly / daily / weekly, compared against this hour /
34
+ # today / this week -- but in a fixed number of queries instead of about six
35
+ # per pair. DashboardStats calls this from the live stats broadcast, which
36
+ # runs inside the host app's capture path.
37
+ #
38
+ # Only pairs with an event this week are looked at: a pair with none has a
39
+ # count of zero, which no baseline can call anomalous.
40
+ #
41
+ # @return [Array<Hash>] error_type, platform, count, level, std_devs_above,
42
+ # baseline_type. Empty on any failure -- never raises.
43
+ def self.current_anomalies(sensitivity: 2, application_id: nil)
44
+ return [] unless defined?(ErrorBaseline) && ErrorBaseline.table_exists?
45
+
46
+ counts = current_counts_by_pair(application_id: application_id)
47
+ return [] if counts.empty?
48
+
49
+ baselines = latest_baselines_for(counts.keys)
50
+
51
+ counts.filter_map do |(error_type, platform), windows|
52
+ kind = BASELINE_PRECEDENCE.find { |type| baselines[[ error_type, platform, type ]] }
53
+ next unless kind
54
+
55
+ baseline = baselines[[ error_type, platform, kind ]]
56
+ # A flat history has no spread to measure against; dividing by it
57
+ # makes any count above the mean infinitely anomalous.
58
+ next if baseline.std_dev.nil? || baseline.std_dev.zero?
59
+
60
+ count = windows.fetch(kind.to_sym)
61
+ level = baseline.anomaly_level(count, sensitivity: sensitivity)
62
+ next unless level
63
+
64
+ {
65
+ error_type: error_type,
66
+ platform: platform,
67
+ count: count,
68
+ level: level,
69
+ std_devs_above: baseline.std_devs_above_mean(count),
70
+ baseline_type: kind
71
+ }
72
+ end
73
+ rescue => e
74
+ RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] current_anomalies failed: #{e.class}: #{e.message}")
75
+ []
76
+ end
77
+
78
+ # { [error_type, platform] => { hourly:, daily:, weekly: } } in ONE grouped
79
+ # query, counted in the units the baselines were built from (occurrence
80
+ # rows when that table exists). The week is the widest window, so it is
81
+ # the WHERE; the day and the hour are conditional sums inside it.
82
+ def self.current_counts_by_pair(application_id: nil)
83
+ logs = ErrorLog.table_name
84
+ column = Services::BaselineCalculator.time_column
85
+ now = Time.current
86
+
87
+ relation = if defined?(ErrorOccurrence) && ErrorOccurrence.table_exists?
88
+ ErrorOccurrence.joins(:error_log)
89
+ else
90
+ ErrorLog.all
91
+ end
92
+ relation = relation.where(logs => { application_id: application_id }) if application_id.present?
93
+
94
+ since = ->(time) { ErrorLog.sanitize_sql_array([ "SUM(CASE WHEN #{column} >= ? THEN 1 ELSE 0 END)", time ]) }
95
+
96
+ rows = relation
97
+ .where("#{column} >= ?", now.beginning_of_week)
98
+ .group("#{logs}.error_type", "#{logs}.platform")
99
+ .order(Arel.sql("COUNT(*) DESC"))
100
+ .limit(MAX_ANOMALY_PAIRS)
101
+ .pluck(Arel.sql("#{logs}.error_type"), Arel.sql("#{logs}.platform"), Arel.sql("COUNT(*)"),
102
+ Arel.sql(since.call(now.beginning_of_day)), Arel.sql(since.call(now.beginning_of_hour)))
103
+
104
+ rows.to_h do |error_type, platform, weekly, daily, hourly|
105
+ [ [ error_type, platform ], { weekly: weekly.to_i, daily: daily.to_i, hourly: hourly.to_i } ]
106
+ end
107
+ end
108
+
109
+ # { [error_type, platform, baseline_type] => ErrorBaseline }, the most recent
110
+ # row of each, in ONE query. Baseline rows accumulate (one per calculation
111
+ # period), so "latest" is resolved in SQL with a join on MAX(period_start)
112
+ # rather than by loading the history. The join form is portable to every
113
+ # adapter; a row-value IN is not.
114
+ def self.latest_baselines_for(pairs)
115
+ table = ErrorBaseline.table_name
116
+ types = pairs.map(&:first).compact.uniq
117
+ return {} if types.empty?
118
+
119
+ latest = ErrorBaseline.where(error_type: types, baseline_type: BASELINE_PRECEDENCE)
120
+ .group(:error_type, :platform, :baseline_type)
121
+ .select(:error_type, :platform, :baseline_type, "MAX(period_start) AS latest_period_start")
122
+
123
+ ErrorBaseline
124
+ .joins("INNER JOIN (#{latest.to_sql}) latest_baselines ON " \
125
+ "latest_baselines.error_type = #{table}.error_type AND " \
126
+ "latest_baselines.platform = #{table}.platform AND " \
127
+ "latest_baselines.baseline_type = #{table}.baseline_type AND " \
128
+ "latest_baselines.latest_period_start = #{table}.period_start")
129
+ .index_by { |baseline| [ baseline.error_type, baseline.platform, baseline.baseline_type ] }
130
+ end
131
+ private_class_method :current_counts_by_pair, :latest_baselines_for
132
+
26
133
  def initialize(error_type, platform)
27
134
  @error_type = error_type
28
135
  @platform = platform