rails_error_dashboard 0.12.0 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (90) hide show
  1. checksums.yaml +4 -4
  2. data/app/controllers/rails_error_dashboard/application_controller.rb +92 -11
  3. data/app/controllers/rails_error_dashboard/errors_controller.rb +111 -34
  4. data/app/controllers/rails_error_dashboard/webhooks_controller.rb +3 -0
  5. data/app/helpers/rails_error_dashboard/application_helper.rb +46 -1
  6. data/app/helpers/rails_error_dashboard/backtrace_helper.rb +8 -0
  7. data/app/jobs/rails_error_dashboard/concerns/plain_channel_message.rb +71 -0
  8. data/app/jobs/rails_error_dashboard/notification_burst_summary_job.rb +55 -0
  9. data/app/jobs/rails_error_dashboard/retention_cleanup_job.rb +68 -1
  10. data/app/jobs/rails_error_dashboard/storm_notification_job.rb +8 -44
  11. data/app/models/rails_error_dashboard/error_baseline.rb +12 -9
  12. data/app/models/rails_error_dashboard/error_comment.rb +0 -5
  13. data/app/models/rails_error_dashboard/error_log.rb +9 -3
  14. data/app/models/rails_error_dashboard/error_logs_record.rb +34 -0
  15. data/app/models/rails_error_dashboard/error_occurrence.rb +9 -1
  16. data/app/views/layouts/rails_error_dashboard.html.erb +11 -3
  17. data/app/views/rails_error_dashboard/errors/_discussion.html.erb +18 -11
  18. data/app/views/rails_error_dashboard/errors/_issue_section.html.erb +22 -8
  19. data/app/views/rails_error_dashboard/errors/_sidebar_metadata.html.erb +15 -6
  20. data/app/views/rails_error_dashboard/errors/analytics.html.erb +14 -8
  21. data/app/views/rails_error_dashboard/errors/diagnostic_dumps.html.erb +1 -1
  22. data/app/views/rails_error_dashboard/errors/index.html.erb +6 -1
  23. data/app/views/rails_error_dashboard/errors/platform_comparison.html.erb +7 -7
  24. data/app/views/rails_error_dashboard/errors/releases.html.erb +2 -2
  25. data/app/views/rails_error_dashboard/errors/settings.html.erb +2 -0
  26. data/config/locales/de.yml +28 -0
  27. data/config/locales/en.yml +35 -0
  28. data/config/locales/es.yml +28 -0
  29. data/config/locales/fr.yml +28 -0
  30. data/config/locales/it.yml +28 -0
  31. data/config/locales/ja.yml +28 -0
  32. data/config/locales/pl.yml +28 -0
  33. data/config/locales/pt-BR.yml +28 -0
  34. data/config/locales/ru.yml +28 -0
  35. data/config/locales/uk.yml +28 -0
  36. data/config/locales/zh-CN.yml +28 -0
  37. data/db/migrate/20260917000001_add_last_notified_at_to_error_logs.rb +40 -0
  38. data/lib/generators/rails_error_dashboard/install/templates/initializer.rb +12 -1
  39. data/lib/rails_error_dashboard/commands/assign_error.rb +12 -2
  40. data/lib/rails_error_dashboard/commands/backfill_environments.rb +2 -0
  41. data/lib/rails_error_dashboard/commands/backfill_resolved_at.rb +42 -0
  42. data/lib/rails_error_dashboard/commands/batch_delete_errors.rb +1 -0
  43. data/lib/rails_error_dashboard/commands/batch_mute_errors.rb +2 -0
  44. data/lib/rails_error_dashboard/commands/batch_resolve_errors.rb +2 -0
  45. data/lib/rails_error_dashboard/commands/batch_unmute_errors.rb +2 -0
  46. data/lib/rails_error_dashboard/commands/find_or_increment_error.rb +39 -7
  47. data/lib/rails_error_dashboard/commands/flush_rack_attack_events.rb +3 -1
  48. data/lib/rails_error_dashboard/commands/flush_storm_counts.rb +27 -4
  49. data/lib/rails_error_dashboard/commands/flush_swallowed_exceptions.rb +5 -2
  50. data/lib/rails_error_dashboard/commands/link_existing_issue.rb +1 -0
  51. data/lib/rails_error_dashboard/commands/log_error.rb +82 -23
  52. data/lib/rails_error_dashboard/commands/mute_error.rb +1 -0
  53. data/lib/rails_error_dashboard/commands/resolve_error.rb +2 -0
  54. data/lib/rails_error_dashboard/commands/scrub_invalid_encoding.rb +104 -0
  55. data/lib/rails_error_dashboard/commands/snooze_error.rb +34 -10
  56. data/lib/rails_error_dashboard/commands/unmute_error.rb +1 -0
  57. data/lib/rails_error_dashboard/commands/update_error_priority.rb +23 -2
  58. data/lib/rails_error_dashboard/commands/update_error_status.rb +33 -6
  59. data/lib/rails_error_dashboard/configuration.rb +19 -1
  60. data/lib/rails_error_dashboard/engine.rb +15 -0
  61. data/lib/rails_error_dashboard/queries/analytics_stats.rb +107 -26
  62. data/lib/rails_error_dashboard/queries/baseline_stats.rb +107 -0
  63. data/lib/rails_error_dashboard/queries/dashboard_stats.rb +55 -50
  64. data/lib/rails_error_dashboard/queries/error_correlation.rb +6 -3
  65. data/lib/rails_error_dashboard/queries/errors_list.rb +15 -2
  66. data/lib/rails_error_dashboard/queries/similar_errors.rb +1 -1
  67. data/lib/rails_error_dashboard/services/analytics_cache_manager.rb +43 -18
  68. data/lib/rails_error_dashboard/services/backtrace_processor.rb +3 -1
  69. data/lib/rails_error_dashboard/services/cause_chain_extractor.rb +3 -1
  70. data/lib/rails_error_dashboard/services/codeberg_issue_client.rb +13 -2
  71. data/lib/rails_error_dashboard/services/diagnostic_dump_generator.rb +5 -3
  72. data/lib/rails_error_dashboard/services/encoding_sanitizer.rb +80 -0
  73. data/lib/rails_error_dashboard/services/error_broadcaster.rb +180 -32
  74. data/lib/rails_error_dashboard/services/error_hash_generator.rb +9 -3
  75. data/lib/rails_error_dashboard/services/error_notification_dispatcher.rb +13 -0
  76. data/lib/rails_error_dashboard/services/exception_filter.rb +55 -0
  77. data/lib/rails_error_dashboard/services/git_head_reader.rb +102 -0
  78. data/lib/rails_error_dashboard/services/notification_throttler.rb +172 -28
  79. data/lib/rails_error_dashboard/services/sensitive_data_filter.rb +47 -1
  80. data/lib/rails_error_dashboard/services/storm_protection/circuit_breaker.rb +59 -3
  81. data/lib/rails_error_dashboard/services/storm_protection/fingerprint_buckets.rb +25 -2
  82. data/lib/rails_error_dashboard/services/storm_protection/gate.rb +32 -5
  83. data/lib/rails_error_dashboard/services/swallowed_exception_tracker.rb +99 -23
  84. data/lib/rails_error_dashboard/services/url_safety.rb +40 -0
  85. data/lib/rails_error_dashboard/subscribers/issue_tracker_subscriber.rb +11 -2
  86. data/lib/rails_error_dashboard/value_objects/error_context.rb +3 -1
  87. data/lib/rails_error_dashboard/version.rb +1 -1
  88. data/lib/rails_error_dashboard.rb +31 -0
  89. data/lib/tasks/error_dashboard.rake +54 -4
  90. metadata +10 -2
@@ -2,29 +2,54 @@
2
2
 
3
3
  module RailsErrorDashboard
4
4
  module Services
5
- # Infrastructure service: Clear analytics caches
5
+ # Infrastructure service: invalidate the dashboard's cached statistics
6
6
  #
7
- # Clears dashboard_stats, analytics_stats, and platform_comparison
8
- # cache entries. Handles cache stores that don't support delete_matched
9
- # (e.g., SolidCache) gracefully.
7
+ # Every cached stats key (DashboardStats, AnalyticsStats) embeds a
8
+ # GENERATION number read from Rails.cache. Invalidation is one increment of
9
+ # that number: entries written under the old generation are simply never
10
+ # read again and age out by their own TTL.
11
+ #
12
+ # This replaced delete_matched("dashboard_stats/*") and friends, which is a
13
+ # SCAN of the HOST app's whole Redis keyspace, NotImplementedError on
14
+ # memcached, and used to run from ErrorLog's after_save -- i.e. inside the
15
+ # host's request thread on every captured error.
16
+ #
17
+ # Who calls .clear: user actions and maintenance that change what the cards
18
+ # show (resolve, status change, batch actions, mute, retention). Captures
19
+ # deliberately do NOT: they rely on the TTL (1 minute for the stat cards,
20
+ # 5 for analytics), which decouples capture cost from dashboard freshness.
21
+ #
22
+ # With :memory_store the generation is per process, so an action handled by
23
+ # one worker reaches the others when their entries expire. With :null_store
24
+ # nothing is cached and there is nothing to invalidate.
10
25
  class AnalyticsCacheManager
11
- CACHE_PATTERNS = %w[
12
- dashboard_stats/*
13
- analytics_stats/*
14
- platform_comparison/*
15
- ].freeze
26
+ GENERATION_KEY = "red/cache_gen"
27
+
28
+ # @return [Integer] the current cache generation; 0 when unset or unreadable
29
+ def self.generation
30
+ Rails.cache.read(GENERATION_KEY).to_i
31
+ rescue => e
32
+ RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] cache generation unreadable: #{e.class}: #{e.message}")
33
+ 0
34
+ end
16
35
 
17
- # Clear all analytics caches
36
+ # Invalidate every cached dashboard statistic. Never raises.
37
+ #
38
+ # A plain read + write, deliberately not Rails.cache.increment: increment
39
+ # is unimplemented on some stores, returns nil for a missing key on
40
+ # others, and on Redis is an INCRBY that fails against an entry written
41
+ # by #write. It does not need to be atomic either -- racing clears only
42
+ # need the number to differ from the one stale entries were written under.
43
+ #
44
+ # Never lower than the clock in milliseconds, so that a store which evicts
45
+ # the generation key cannot restart at a number that entries still inside
46
+ # their TTL were written under a few minutes earlier.
18
47
  def self.clear
19
- if Rails.cache.respond_to?(:delete_matched)
20
- CACHE_PATTERNS.each { |pattern| Rails.cache.delete_matched(pattern) }
21
- else
22
- Rails.logger.info("Cache store doesn't support delete_matched, skipping cache clear") if Rails.logger
23
- end
24
- rescue NotImplementedError => e
25
- Rails.logger.info("Cache store doesn't support delete_matched: #{e.message}") if Rails.logger
48
+ Rails.cache.write(GENERATION_KEY, [ generation + 1, (Time.now.to_f * 1000).to_i ].max)
49
+ nil
26
50
  rescue => e
27
- Rails.logger.error("Failed to clear analytics cache: #{e.message}") if Rails.logger
51
+ RailsErrorDashboard::Logger.error("[RailsErrorDashboard] Failed to clear analytics cache: #{e.class}: #{e.message}")
52
+ nil
28
53
  end
29
54
  end
30
55
  end
@@ -22,7 +22,9 @@ module RailsErrorDashboard
22
22
 
23
23
  max_lines ||= RailsErrorDashboard.configuration.max_backtrace_lines
24
24
 
25
- limited_backtrace = backtrace.first(max_lines).map { |line| shorten_gem_path(line) }
25
+ # Scrubbed per line: a frame label can hold any bytes, and both the sub
26
+ # below and the join (mixed encodings) raise on an invalid one.
27
+ limited_backtrace = backtrace.first(max_lines).map { |line| shorten_gem_path(EncodingSanitizer.scrub(line)) }
26
28
  result = limited_backtrace.join("\n")
27
29
 
28
30
  if backtrace.length > max_lines
@@ -43,7 +43,9 @@ module RailsErrorDashboard
43
43
  depth += 1
44
44
  end
45
45
 
46
- chain.empty? ? nil : chain.to_json
46
+ # to_json raises on a cause message or frame with invalid bytes, which
47
+ # used to drop the whole chain.
48
+ chain.empty? ? nil : EncodingSanitizer.scrub_deep(chain).to_json
47
49
  rescue => e
48
50
  # SAFETY: Never let cause chain extraction break error logging
49
51
  RailsErrorDashboard::Logger.debug(
@@ -40,7 +40,7 @@ module RailsErrorDashboard
40
40
  auth_headers
41
41
  )
42
42
 
43
- response[:status] == 201 ? success_response({}) : error_response("Codeberg API error (#{response[:status]})")
43
+ patch_result(response)
44
44
  end
45
45
 
46
46
  def reopen_issue(number:)
@@ -50,7 +50,7 @@ module RailsErrorDashboard
50
50
  auth_headers
51
51
  )
52
52
 
53
- response[:status] == 201 ? success_response({}) : error_response("Codeberg API error (#{response[:status]})")
53
+ patch_result(response)
54
54
  end
55
55
 
56
56
  def add_comment(number:, body:)
@@ -114,6 +114,17 @@ module RailsErrorDashboard
114
114
 
115
115
  private
116
116
 
117
+ # Gitea/Forgejo answer a successful PATCH with 200; 201 is what they
118
+ # return for creation. Accepting only 201 reported every close and reopen
119
+ # as failed although the forge had applied it.
120
+ def patch_result(response)
121
+ if [ 200, 201 ].include?(response[:status])
122
+ success_response({})
123
+ else
124
+ error_response("Codeberg API error (#{response[:status]})")
125
+ end
126
+ end
127
+
117
128
  def auth_headers
118
129
  { "Authorization" => "token #{@token}" }
119
130
  end
@@ -27,7 +27,9 @@ module RailsErrorDashboard
27
27
  end
28
28
 
29
29
  def call
30
- {
30
+ # Thread names and breadcrumb text are arbitrary bytes; both callers
31
+ # immediately to_json this, which raises on an invalid one.
32
+ EncodingSanitizer.scrub_deep(
31
33
  captured_at: Time.current.iso8601,
32
34
  pid: Process.pid,
33
35
  uptime_seconds: process_uptime,
@@ -37,9 +39,9 @@ module RailsErrorDashboard
37
39
  threads: thread_info,
38
40
  gc: gc_info,
39
41
  object_counts: object_counts
40
- }
42
+ )
41
43
  rescue => e
42
- { captured_at: Time.current.iso8601, error: e.message }
44
+ { captured_at: Time.current.iso8601, error: EncodingSanitizer.scrub(e.message) }
43
45
  end
44
46
 
45
47
  private
@@ -0,0 +1,80 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RailsErrorDashboard
4
+ module Services
5
+ # Pure algorithm: make every String in a value safe to store, serialize and render
6
+ #
7
+ # Exception messages, backtrace lines, request URLs, user agents and params can
8
+ # carry bytes that are not valid UTF-8 (a binary upload echoed in a message, a
9
+ # Latin-1 query string, a NUL from a fuzzer). Left alone they raise at the
10
+ # worst possible moments: ActiveJob JSON-encoding the async payload in the
11
+ # host's request thread, the PostgreSQL INSERT, or `blank?` while rendering the
12
+ # error's own page.
13
+ #
14
+ # Invalid sequences become "?" and NUL bytes are removed (PostgreSQL text
15
+ # columns reject them). Strings tagged ASCII-8BIT (or any other encoding) are
16
+ # re-tagged UTF-8 and then scrubbed — not transcoded, because a binary tag
17
+ # says nothing about what the bytes mean and transcoding would raise.
18
+ #
19
+ # The common case costs one `valid_encoding?` scan and returns the SAME
20
+ # object. Nothing here raises, and the caller's objects are never mutated.
21
+ #
22
+ # @example
23
+ # EncodingSanitizer.scrub("caf\xC3 \xFF".b) # => "caf? ?"
24
+ # EncodingSanitizer.scrub_deep({ url: "/?q=\xFF" }) # => { url: "/?q=?" }
25
+ class EncodingSanitizer
26
+ REPLACEMENT = "?"
27
+ UNREADABLE = "[unreadable]"
28
+ TOO_DEEP = "[truncated: nested too deep]"
29
+ MAX_DEPTH = 10
30
+ NUL = "\0"
31
+
32
+ # @param str [Object] anything; only Strings are touched
33
+ # @return [Object] the same object when already clean, otherwise a clean copy
34
+ def self.scrub(str)
35
+ return str unless String === str
36
+
37
+ s = str.encoding == Encoding::UTF_8 ? str : str.dup.force_encoding(Encoding::UTF_8)
38
+ return s if s.valid_encoding? && !s.include?(NUL)
39
+
40
+ s = s.scrub(REPLACEMENT) unless s.valid_encoding?
41
+ s.delete(NUL)
42
+ rescue => e
43
+ RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] EncodingSanitizer.scrub failed: #{e.class}")
44
+ UNREADABLE
45
+ end
46
+
47
+ # Recursively scrub Strings inside Hashes (keys and values) and Arrays.
48
+ # Anything else is returned untouched. A container nested deeper than
49
+ # MAX_DEPTH is replaced by a placeholder rather than passed through: an
50
+ # unscrubbed subtree would defeat the point, and a self-referencing one
51
+ # would never end.
52
+ #
53
+ # @param obj [Object]
54
+ # @return [Object]
55
+ def self.scrub_deep(obj, depth = 0)
56
+ case obj
57
+ when String
58
+ scrub(obj)
59
+ when Hash
60
+ return TOO_DEEP if depth >= MAX_DEPTH
61
+
62
+ # Build into an empty copy of the same class so a
63
+ # HashWithIndifferentAccess stays one.
64
+ obj.each_with_object(obj.class.new) do |(key, value), result|
65
+ result[scrub_deep(key, depth + 1)] = scrub_deep(value, depth + 1)
66
+ end
67
+ when Array
68
+ return TOO_DEEP if depth >= MAX_DEPTH
69
+
70
+ obj.map { |value| scrub_deep(value, depth + 1) }
71
+ else
72
+ obj
73
+ end
74
+ rescue => e
75
+ RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] EncodingSanitizer.scrub_deep failed: #{e.class}")
76
+ String === obj ? UNREADABLE : obj
77
+ end
78
+ end
79
+ end
80
+ end
@@ -15,26 +15,71 @@ module RailsErrorDashboard
15
15
  # are NOT available there. We render via the engine's own controller renderer
16
16
  # and pass pre-rendered HTML to the broadcast to ensure route helpers work.
17
17
  class ErrorBroadcaster
18
+ # Minimum seconds between two stats broadcasts on one stream, per process.
19
+ # The row broadcasts are cheap (one partial); the stats payload is a full
20
+ # DashboardStats computation, and a capture runs it from inside the host
21
+ # app's request thread. Leading edge: the first event in a window
22
+ # broadcasts, later ones are dropped and corrected by the next window or
23
+ # the next page load.
24
+ STATS_BROADCAST_INTERVAL = 5
25
+
26
+ # Stream names, in ONE place for the view (turbo_stream_from) and for this
27
+ # class. Three kinds, because a Turbo subscription is all-or-nothing per
28
+ # stream and the index needs them separately: a view filtered by platform
29
+ # wants row REPLACES (harmless: they only touch rows already on the page)
30
+ # but not PREPENDS (a new row may not match its filter).
31
+ #
32
+ # Every kind exists globally and per application ("error_list_app_7"), so
33
+ # a view filtered to one application never receives another's rows.
34
+ STREAMS = { list: "error_list", updates: "error_updates", stats: "error_stats" }.freeze
35
+
36
+ # Column visibility (which optional cells a row has) used to be two
37
+ # SELECT DISTINCT scans per broadcast. It changes about never.
38
+ COLUMN_VISIBILITY_TTL = 60.seconds
39
+
40
+ # One throttle entry per stats stream, i.e. per application. Bounded so a
41
+ # host with thousands of applications cannot grow it without limit.
42
+ MAX_THROTTLED_STREAMS = 200
43
+
44
+ THROTTLE_MUTEX = Mutex.new
45
+ private_constant :THROTTLE_MUTEX
46
+
47
+ # @param kind [Symbol] :list (new rows), :updates (row replaces) or :stats
48
+ # @param application_id [Integer, String, nil] nil/blank for the global stream
49
+ # @return [String]
50
+ def self.stream_name(kind, application_id = nil)
51
+ base = STREAMS.fetch(kind)
52
+ return base if application_id.blank?
53
+
54
+ # to_i: the id reaches here from a request parameter in the view.
55
+ "#{base}_app_#{application_id.to_s[/\A\d+\z/].to_i}"
56
+ end
57
+
58
+ # The streams a dashboard index page subscribes to.
59
+ # @param application_id [Integer, String, nil] the page's application filter
60
+ # @param filtered [Boolean] any OTHER filter, a sort or a later page is active,
61
+ # so a new row cannot simply be put on top of the list
62
+ # @return [Array<String>]
63
+ def self.streams_for_view(application_id: nil, filtered: false)
64
+ kinds = filtered ? %i[updates stats] : %i[list updates stats]
65
+ kinds.map { |kind| stream_name(kind, application_id) }
66
+ end
67
+
18
68
  # Broadcast a new error (prepend to error list + refresh stats)
19
69
  # @param error_log [ErrorLog] The newly created error
20
70
  def self.broadcast_new(error_log)
21
71
  return unless error_log
22
72
  return unless available?
73
+ return unless calm?
23
74
 
24
- platforms = ErrorLog.distinct.pluck(:platform).compact
25
- show_platform = platforms.size > 1
26
- show_environment = ErrorLog.column_names.include?("environment") &&
27
- ErrorLog.distinct.pluck(:environment).compact.size > 1
28
-
29
- html = render_partial("rails_error_dashboard/errors/error_row",
30
- error: error_log, show_platform: show_platform, show_environment: show_environment)
31
-
32
- Turbo::StreamsChannel.broadcast_prepend_to(
33
- "error_list",
34
- target: "error_list",
35
- html: html
36
- )
37
- broadcast_stats
75
+ each_row_html(error_log) do |application_id, html|
76
+ Turbo::StreamsChannel.broadcast_prepend_to(
77
+ stream_name(:list, application_id),
78
+ target: "error_list",
79
+ html: html
80
+ )
81
+ end
82
+ broadcast_stats(error_log.application_id)
38
83
  rescue => e
39
84
  Rails.logger.error("[RailsErrorDashboard] Failed to broadcast new error: #{e.class} - #{e.message}")
40
85
  Rails.logger.debug("[RailsErrorDashboard] Backtrace: #{e.backtrace&.first(3)&.join("\n")}")
@@ -45,37 +90,45 @@ module RailsErrorDashboard
45
90
  def self.broadcast_update(error_log)
46
91
  return unless error_log
47
92
  return unless available?
93
+ return unless calm?
48
94
 
49
- platforms = ErrorLog.distinct.pluck(:platform).compact
50
- show_platform = platforms.size > 1
51
- show_environment = ErrorLog.column_names.include?("environment") &&
52
- ErrorLog.distinct.pluck(:environment).compact.size > 1
53
-
54
- html = render_partial("rails_error_dashboard/errors/error_row",
55
- error: error_log, show_platform: show_platform, show_environment: show_environment)
56
-
57
- Turbo::StreamsChannel.broadcast_replace_to(
58
- "error_list",
59
- target: "error_#{error_log.id}",
60
- html: html
61
- )
62
- broadcast_stats
95
+ each_row_html(error_log) do |application_id, html|
96
+ Turbo::StreamsChannel.broadcast_replace_to(
97
+ stream_name(:updates, application_id),
98
+ target: "error_#{error_log.id}",
99
+ html: html
100
+ )
101
+ end
102
+ broadcast_stats(error_log.application_id)
63
103
  rescue => e
64
104
  Rails.logger.error("[RailsErrorDashboard] Failed to broadcast error update: #{e.class} - #{e.message}")
65
105
  Rails.logger.debug("[RailsErrorDashboard] Backtrace: #{e.backtrace&.first(3)&.join("\n")}")
66
106
  end
67
107
 
68
- # Broadcast stats refresh
69
- def self.broadcast_stats
108
+ # Broadcast a stats refresh: all-application stats to the global stream,
109
+ # application-scoped stats to that application's stream.
110
+ #
111
+ # ONE event computes at most ONE payload. Each stream is throttled on its
112
+ # own (STATS_BROADCAST_INTERVAL); when both are due, the one that has
113
+ # waited longest goes and the other is served by the next event. Computing
114
+ # both would double the most expensive thing a capture does, and a
115
+ # fixed order would starve the second stream whenever events are rare.
116
+ #
117
+ # @param application_id [Integer, nil] the application of the event, if any
118
+ def self.broadcast_stats(application_id = nil)
70
119
  return unless available?
120
+ return unless calm?
71
121
 
72
- stats = Queries::DashboardStats.call
122
+ scope = claim_stats_window!([ nil, application_id.presence ].uniq)
123
+ return if scope == :none
124
+
125
+ stats = scope ? Queries::DashboardStats.call(application_id: scope) : Queries::DashboardStats.call
73
126
  return unless stats.is_a?(Hash) && stats.present?
74
127
 
75
128
  html = render_partial("rails_error_dashboard/errors/stats", stats: stats)
76
129
 
77
130
  Turbo::StreamsChannel.broadcast_replace_to(
78
- "error_list",
131
+ stream_name(:stats, scope),
79
132
  target: "dashboard_stats",
80
133
  html: html
81
134
  )
@@ -84,6 +137,101 @@ module RailsErrorDashboard
84
137
  Rails.logger.debug("[RailsErrorDashboard] Backtrace: #{e.backtrace&.first(3)&.join("\n")}")
85
138
  end
86
139
 
140
+ # Yields [application_id_or_nil, row_html] once per row stream. The global
141
+ # table has an Application column when there is more than one application;
142
+ # an application's own table never does, so the two need different cells.
143
+ # Rendered once when they come out the same.
144
+ def self.each_row_html(error_log)
145
+ render = lambda do |visibility, show_application|
146
+ render_partial("rails_error_dashboard/errors/error_row",
147
+ error: error_log, show_platform: visibility[:platform],
148
+ show_environment: visibility[:environment], show_application: show_application)
149
+ end
150
+
151
+ global = column_visibility(nil)
152
+ yield nil, render.call(global, global[:application])
153
+
154
+ application_id = error_log.application_id
155
+ return if application_id.blank?
156
+
157
+ yield application_id, render.call(column_visibility(application_id), false)
158
+ end
159
+
160
+ # Which optional columns the index table shows, for the global view or for
161
+ # one application -- the same rule the index itself applies. Cached for
162
+ # COLUMN_VISIBILITY_TTL and keyed on the cache generation, so a user
163
+ # action that could change it (a batch delete) refreshes it at once.
164
+ def self.column_visibility(application_id)
165
+ key = [ "red/columns", AnalyticsCacheManager.generation, application_id || "all" ].join("/")
166
+
167
+ Rails.cache.fetch(key, expires_in: COLUMN_VISIBILITY_TTL) do
168
+ scope = application_id ? ErrorLog.where(application_id: application_id) : ErrorLog.all
169
+ {
170
+ platform: scope.distinct.pluck(:platform).compact.size > 1,
171
+ environment: ErrorLog.column_names.include?("environment") &&
172
+ scope.distinct.pluck(:environment).compact.size > 1,
173
+ application: application_id.nil? && Application.count > 1
174
+ }
175
+ end
176
+ end
177
+
178
+ # Live updates are a convenience; during a storm they are load. While the
179
+ # breaker is anything but :closed nothing is broadcast -- the page catches
180
+ # up on its next load. Fails towards broadcasting: Gate.state itself
181
+ # answers :closed when storm protection is off or unreadable.
182
+ def self.calm?
183
+ StormProtection::Gate.state == :closed
184
+ rescue StandardError
185
+ true
186
+ end
187
+
188
+ # Picks which stats scope this event may broadcast, and claims its window.
189
+ #
190
+ # @param scopes [Array<Integer, nil>] candidate scopes (nil = all applications)
191
+ # @return [Integer, nil, :none] the claimed scope, or :none when every
192
+ # candidate is still inside its window
193
+ #
194
+ # The claim is taken BEFORE the stats are computed, so concurrent captures
195
+ # cannot all decide to compute at once, and a computation that fails does
196
+ # not get retried by every event that follows it.
197
+ def self.claim_stats_window!(scopes = [ nil ])
198
+ THROTTLE_MUTEX.synchronize do
199
+ @stats_claimed_at ||= {}
200
+ now = monotonic_now
201
+
202
+ due = scopes.select do |scope|
203
+ last = @stats_claimed_at[scope]
204
+ last.nil? || (now - last) >= STATS_BROADCAST_INTERVAL
205
+ end
206
+ return :none if due.empty?
207
+
208
+ # Longest-waiting first; never-claimed counts as waiting forever.
209
+ # min_by is stable, so the global stream wins a tie.
210
+ chosen = due.min_by { |scope| @stats_claimed_at[scope] || -Float::INFINITY }
211
+ @stats_claimed_at[chosen] = now
212
+
213
+ if @stats_claimed_at.size > MAX_THROTTLED_STREAMS
214
+ stalest = @stats_claimed_at.min_by { |_scope, at| at }.first
215
+ @stats_claimed_at.delete(stalest)
216
+ end
217
+
218
+ chosen
219
+ end
220
+ end
221
+
222
+ def self.throttled_stream_count
223
+ THROTTLE_MUTEX.synchronize { (@stats_claimed_at || {}).size }
224
+ end
225
+
226
+ def self.monotonic_now
227
+ Process.clock_gettime(Process::CLOCK_MONOTONIC)
228
+ end
229
+
230
+ # For specs and for a forked worker that wants a clean slate.
231
+ def self.reset_throttle!
232
+ THROTTLE_MUTEX.synchronize { @stats_claimed_at = {} }
233
+ end
234
+
87
235
  # Render a partial using the engine's controller renderer.
88
236
  # This ensures engine route helpers (error_path, etc.) are available,
89
237
  # unlike Turbo's default ApplicationController.render which uses the host app's context.
@@ -121,8 +121,13 @@ module RailsErrorDashboard
121
121
  # @param message [String, nil] The error message
122
122
  # @return [String, nil] Normalized message
123
123
  def self.normalize_message(message)
124
- message
125
- &.[](0, HASH_MESSAGE_LIMIT)
124
+ # Slice, THEN scrub, then match: every path (sync, async, storm gate)
125
+ # goes through here, so they all hash the same text. gsub raises on a
126
+ # string with invalid bytes; a valid message is returned by scrub as-is,
127
+ # so no existing fingerprint changes.
128
+ prefix = message&.[](0, HASH_MESSAGE_LIMIT)
129
+
130
+ EncodingSanitizer.scrub(prefix)
126
131
  &.gsub(/0x[0-9a-f]+/i, "HEX") # Replace hex addresses (before numbers)
127
132
  &.gsub(/#<[^>]+>/, "#<OBJ>") # Replace object inspections
128
133
  &.gsub(/\d+/, "N") # Replace numbers
@@ -161,7 +166,8 @@ module RailsErrorDashboard
161
166
  !frame.include?("/gems/")
162
167
  }
163
168
 
164
- first_app_frame&.split(":")&.first
169
+ # split raises on a frame with invalid bytes (a method name can hold any).
170
+ EncodingSanitizer.scrub(first_app_frame)&.split(":")&.first
165
171
  end
166
172
 
167
173
  # Try custom fingerprint lambda if configured
@@ -10,6 +10,19 @@ module RailsErrorDashboard
10
10
  # @example
11
11
  # ErrorNotificationDispatcher.call(error_log)
12
12
  class ErrorNotificationDispatcher
13
+ # Is any channel configured to the point where .call would enqueue
14
+ # something? The same conditions as below, without the side effects.
15
+ # @return [Boolean]
16
+ def self.any_channel?
17
+ config = RailsErrorDashboard.configuration
18
+
19
+ (config.enable_slack_notifications && config.slack_webhook_url.present?) ||
20
+ (config.enable_email_notifications && config.notification_email_recipients.present?) ||
21
+ (config.enable_discord_notifications && config.discord_webhook_url.present?) ||
22
+ (config.enable_pagerduty_notifications && config.pagerduty_integration_key.present?) ||
23
+ (config.enable_webhook_notifications && config.webhook_urls.present?) || false
24
+ end
25
+
13
26
  # @param error_log [ErrorLog] The error to notify about
14
27
  def self.call(error_log)
15
28
  # OTel: emit a child span around the dispatch so operators can see
@@ -10,6 +10,14 @@ module RailsErrorDashboard
10
10
  # @example
11
11
  # ExceptionFilter.should_log?(exception) # => true/false
12
12
  class ExceptionFilter
13
+ # Distinct errors remembered for first-seen admission (see sampled_out?).
14
+ # Eviction only costs one extra admitted sample, so there is no overflow
15
+ # accounting here.
16
+ MAX_SEEN = 1_000
17
+
18
+ @seen = {}
19
+ @seen_mutex = Mutex.new
20
+
13
21
  # Check if an exception should be logged (not ignored, not sampled out)
14
22
  # @param exception [Exception] The exception to check
15
23
  # @return [Boolean] true if the exception should be logged
@@ -48,16 +56,63 @@ module RailsErrorDashboard
48
56
  # Critical errors are ALWAYS logged regardless of sampling
49
57
  # @param exception [Exception] The exception to check
50
58
  # @return [Boolean] true if the exception should be skipped
59
+ #
60
+ # The FIRST event of each error in this process is always admitted.
61
+ # Sampling is a dice roll taken before anything is known about the error,
62
+ # so at a rate of 0.01 an error that happens three times is usually never
63
+ # recorded at all. Sampling is there to cut volume; it must not hide that
64
+ # an error exists. A rate of 0.0 stays a hard off switch.
51
65
  def self.sampled_out?(exception)
52
66
  sampling_rate = RailsErrorDashboard.configuration.sampling_rate
53
67
 
54
68
  return false if sampling_rate >= 1.0
55
69
  return false if critical?(exception)
56
70
  return true if sampling_rate <= 0.0
71
+ return false if first_sighting?(exception)
57
72
 
58
73
  rand > sampling_rate
59
74
  end
60
75
 
76
+ # True exactly once per process for each (exception class, first
77
+ # application frame). Coarser than the real fingerprint on purpose: the
78
+ # fingerprint needs the application id (a query) and this runs before the
79
+ # storm gate. Two errors raised from one line share a "first".
80
+ #
81
+ # Per process, so N workers admit N firsts. Bounded at MAX_SEEN, least
82
+ # recently seen evicted first. Any failure means "not seen": admit.
83
+ def self.first_sighting?(exception)
84
+ key = seen_key(exception)
85
+
86
+ @seen_mutex.synchronize do
87
+ seen_before = !@seen.delete(key).nil?
88
+ @seen.shift while @seen.size >= MAX_SEEN
89
+ @seen[key] = true
90
+ !seen_before
91
+ end
92
+ rescue => e
93
+ RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] ExceptionFilter.first_sighting? failed: #{e.class}: #{e.message}")
94
+ true
95
+ end
96
+
97
+ # "ClassName|first application frame". The first backtrace line that is
98
+ # not inside a gem or the Ruby core; the first line when there is no such
99
+ # frame; nothing when the exception was never raised.
100
+ def self.seen_key(exception)
101
+ backtrace = exception.backtrace || []
102
+ frame = backtrace.find { |line| !line.include?("/gems/") && !line.start_with?("<internal:") } || backtrace.first
103
+
104
+ "#{exception.class.name}|#{frame.to_s.sub(/:in .*\z/, "")}"
105
+ end
106
+
107
+ def self.seen_size
108
+ @seen_mutex.synchronize { @seen.size }
109
+ end
110
+
111
+ # Forget every sighting (specs, and after a fork).
112
+ def self.reset_seen!
113
+ @seen_mutex.synchronize { @seen.clear }
114
+ end
115
+
61
116
  # Check if exception is a critical error type
62
117
  # @param exception [Exception] The exception to check
63
118
  # @return [Boolean] true if the exception is critical