rails_error_dashboard 0.12.1 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (90) hide show
  1. checksums.yaml +4 -4
  2. data/app/controllers/rails_error_dashboard/application_controller.rb +92 -11
  3. data/app/controllers/rails_error_dashboard/errors_controller.rb +111 -34
  4. data/app/controllers/rails_error_dashboard/webhooks_controller.rb +3 -0
  5. data/app/helpers/rails_error_dashboard/application_helper.rb +46 -1
  6. data/app/helpers/rails_error_dashboard/backtrace_helper.rb +8 -0
  7. data/app/jobs/rails_error_dashboard/concerns/plain_channel_message.rb +71 -0
  8. data/app/jobs/rails_error_dashboard/notification_burst_summary_job.rb +55 -0
  9. data/app/jobs/rails_error_dashboard/retention_cleanup_job.rb +68 -1
  10. data/app/jobs/rails_error_dashboard/storm_notification_job.rb +8 -44
  11. data/app/models/rails_error_dashboard/error_baseline.rb +12 -9
  12. data/app/models/rails_error_dashboard/error_comment.rb +0 -5
  13. data/app/models/rails_error_dashboard/error_log.rb +9 -3
  14. data/app/models/rails_error_dashboard/error_logs_record.rb +34 -0
  15. data/app/models/rails_error_dashboard/error_occurrence.rb +9 -1
  16. data/app/views/layouts/rails_error_dashboard.html.erb +11 -3
  17. data/app/views/rails_error_dashboard/errors/_discussion.html.erb +18 -11
  18. data/app/views/rails_error_dashboard/errors/_issue_section.html.erb +22 -8
  19. data/app/views/rails_error_dashboard/errors/_sidebar_metadata.html.erb +15 -6
  20. data/app/views/rails_error_dashboard/errors/analytics.html.erb +8 -8
  21. data/app/views/rails_error_dashboard/errors/diagnostic_dumps.html.erb +1 -1
  22. data/app/views/rails_error_dashboard/errors/index.html.erb +6 -1
  23. data/app/views/rails_error_dashboard/errors/platform_comparison.html.erb +7 -7
  24. data/app/views/rails_error_dashboard/errors/releases.html.erb +2 -2
  25. data/app/views/rails_error_dashboard/errors/settings.html.erb +2 -0
  26. data/config/locales/de.yml +28 -0
  27. data/config/locales/en.yml +35 -0
  28. data/config/locales/es.yml +28 -0
  29. data/config/locales/fr.yml +28 -0
  30. data/config/locales/it.yml +28 -0
  31. data/config/locales/ja.yml +28 -0
  32. data/config/locales/pl.yml +28 -0
  33. data/config/locales/pt-BR.yml +28 -0
  34. data/config/locales/ru.yml +28 -0
  35. data/config/locales/uk.yml +28 -0
  36. data/config/locales/zh-CN.yml +28 -0
  37. data/db/migrate/20260917000001_add_last_notified_at_to_error_logs.rb +40 -0
  38. data/lib/generators/rails_error_dashboard/install/templates/initializer.rb +12 -1
  39. data/lib/rails_error_dashboard/commands/assign_error.rb +12 -2
  40. data/lib/rails_error_dashboard/commands/backfill_environments.rb +2 -0
  41. data/lib/rails_error_dashboard/commands/backfill_resolved_at.rb +42 -0
  42. data/lib/rails_error_dashboard/commands/batch_delete_errors.rb +1 -0
  43. data/lib/rails_error_dashboard/commands/batch_mute_errors.rb +2 -0
  44. data/lib/rails_error_dashboard/commands/batch_resolve_errors.rb +2 -0
  45. data/lib/rails_error_dashboard/commands/batch_unmute_errors.rb +2 -0
  46. data/lib/rails_error_dashboard/commands/find_or_increment_error.rb +39 -7
  47. data/lib/rails_error_dashboard/commands/flush_rack_attack_events.rb +3 -1
  48. data/lib/rails_error_dashboard/commands/flush_storm_counts.rb +27 -4
  49. data/lib/rails_error_dashboard/commands/flush_swallowed_exceptions.rb +5 -2
  50. data/lib/rails_error_dashboard/commands/link_existing_issue.rb +1 -0
  51. data/lib/rails_error_dashboard/commands/log_error.rb +82 -23
  52. data/lib/rails_error_dashboard/commands/mute_error.rb +1 -0
  53. data/lib/rails_error_dashboard/commands/resolve_error.rb +2 -0
  54. data/lib/rails_error_dashboard/commands/scrub_invalid_encoding.rb +104 -0
  55. data/lib/rails_error_dashboard/commands/snooze_error.rb +34 -10
  56. data/lib/rails_error_dashboard/commands/unmute_error.rb +1 -0
  57. data/lib/rails_error_dashboard/commands/update_error_priority.rb +23 -2
  58. data/lib/rails_error_dashboard/commands/update_error_status.rb +33 -6
  59. data/lib/rails_error_dashboard/configuration.rb +19 -1
  60. data/lib/rails_error_dashboard/engine.rb +15 -0
  61. data/lib/rails_error_dashboard/queries/analytics_stats.rb +4 -3
  62. data/lib/rails_error_dashboard/queries/baseline_stats.rb +107 -0
  63. data/lib/rails_error_dashboard/queries/dashboard_stats.rb +55 -50
  64. data/lib/rails_error_dashboard/queries/error_correlation.rb +6 -3
  65. data/lib/rails_error_dashboard/queries/errors_list.rb +15 -2
  66. data/lib/rails_error_dashboard/queries/similar_errors.rb +1 -1
  67. data/lib/rails_error_dashboard/services/analytics_cache_manager.rb +43 -18
  68. data/lib/rails_error_dashboard/services/backtrace_processor.rb +3 -1
  69. data/lib/rails_error_dashboard/services/cause_chain_extractor.rb +3 -1
  70. data/lib/rails_error_dashboard/services/codeberg_issue_client.rb +13 -2
  71. data/lib/rails_error_dashboard/services/diagnostic_dump_generator.rb +5 -3
  72. data/lib/rails_error_dashboard/services/encoding_sanitizer.rb +80 -0
  73. data/lib/rails_error_dashboard/services/error_broadcaster.rb +180 -32
  74. data/lib/rails_error_dashboard/services/error_hash_generator.rb +9 -3
  75. data/lib/rails_error_dashboard/services/error_notification_dispatcher.rb +13 -0
  76. data/lib/rails_error_dashboard/services/exception_filter.rb +55 -0
  77. data/lib/rails_error_dashboard/services/git_head_reader.rb +102 -0
  78. data/lib/rails_error_dashboard/services/notification_throttler.rb +172 -28
  79. data/lib/rails_error_dashboard/services/sensitive_data_filter.rb +47 -1
  80. data/lib/rails_error_dashboard/services/storm_protection/circuit_breaker.rb +59 -3
  81. data/lib/rails_error_dashboard/services/storm_protection/fingerprint_buckets.rb +25 -2
  82. data/lib/rails_error_dashboard/services/storm_protection/gate.rb +32 -5
  83. data/lib/rails_error_dashboard/services/swallowed_exception_tracker.rb +99 -23
  84. data/lib/rails_error_dashboard/services/url_safety.rb +40 -0
  85. data/lib/rails_error_dashboard/subscribers/issue_tracker_subscriber.rb +11 -2
  86. data/lib/rails_error_dashboard/value_objects/error_context.rb +3 -1
  87. data/lib/rails_error_dashboard/version.rb +1 -1
  88. data/lib/rails_error_dashboard.rb +31 -0
  89. data/lib/tasks/error_dashboard.rake +54 -4
  90. metadata +10 -2
@@ -25,8 +25,11 @@ module RailsErrorDashboard
25
25
  app_id = current_application_id
26
26
 
27
27
  # Process raise counts
28
+ # Keys are scrubbed before they are split: a class name or path with an
29
+ # invalid byte makes split/blank? raise, and the outer rescue would then
30
+ # drop every remaining count in the batch along with it.
28
31
  @raise_counts.each do |key, count|
29
- class_name, location = key.split("|", 2)
32
+ class_name, location = Services::EncodingSanitizer.scrub(key.to_s).split("|", 2)
30
33
  next if class_name.blank? || location.blank?
31
34
 
32
35
  upsert_raise(class_name, location, period, app_id, count)
@@ -34,7 +37,7 @@ module RailsErrorDashboard
34
37
 
35
38
  # Process rescue counts
36
39
  @rescue_counts.each do |key, count|
37
- class_name, locations = key.split("|", 2)
40
+ class_name, locations = Services::EncodingSanitizer.scrub(key.to_s).split("|", 2)
38
41
  next if class_name.blank? || locations.blank?
39
42
 
40
43
  raise_loc, rescue_loc = locations.split("->", 2)
@@ -32,6 +32,7 @@ module RailsErrorDashboard
32
32
 
33
33
  def call
34
34
  return { success: false, error: red_t("red.commands.issue.url_required") } if @issue_url.blank?
35
+ return { success: false, error: red_t("red.commands.issue.url_invalid") } unless Services::UrlSafety.http_url?(@issue_url)
35
36
 
36
37
  error = ErrorLog.find(@error_id)
37
38
  parsed = parse_issue_url(@issue_url)
@@ -106,12 +106,18 @@ module RailsErrorDashboard
106
106
  # FlushStormCounts#canonical_hash does for storm counts.
107
107
  identity_parts = capture_identity_parts(exception, context)
108
108
 
109
- exception_data = {
109
+ # Scrub BEFORE anything is serialized: ActiveJob JSON-encodes the
110
+ # payload on this (the request) thread, and an invalid byte in the
111
+ # message, a backtrace line or the context would raise right here. The
112
+ # identity above is still taken from the raw exception, exactly as the
113
+ # sync path takes it, so grouping is unchanged.
114
+ context = Services::EncodingSanitizer.scrub_deep(context)
115
+ exception_data = Services::EncodingSanitizer.scrub_deep(
110
116
  class_name: exception.class.name,
111
117
  message: exception.message,
112
118
  backtrace: exception.backtrace,
113
119
  cause_chain: serialize_cause_chain(exception)
114
- }
120
+ )
115
121
 
116
122
  # Redact BEFORE the payload crosses the queue boundary. Until now the
117
123
  # filter ran only just before the INSERT, so a durable adapter
@@ -264,6 +270,10 @@ module RailsErrorDashboard
264
270
 
265
271
  context = context.merge(request_params: filtered[:request_params]) if context.key?(:request_params)
266
272
  context = context.merge(request_url: filtered[:request_url]) if context.key?(:request_url)
273
+ # The raw session ID must not sit in Redis / Solid Queue either.
274
+ if context[:session_id]
275
+ context = context.merge(session_id: Services::SensitiveDataFilter.digest_session_id(context[:session_id]))
276
+ end
267
277
 
268
278
  [ exception_data, context ]
269
279
  rescue => e
@@ -292,7 +302,7 @@ module RailsErrorDashboard
292
302
  chain << {
293
303
  class_name: current.class.name,
294
304
  message: current.message&.to_s,
295
- backtrace: current.backtrace&.first(20)&.map { |line| Services::BacktraceProcessor.shorten_gem_path(line) }
305
+ backtrace: current.backtrace&.first(20)&.map { |line| Services::BacktraceProcessor.shorten_gem_path(Services::EncodingSanitizer.scrub(line)) }
296
306
  }
297
307
 
298
308
  current = current.respond_to?(:cause) ? current.cause : nil
@@ -317,7 +327,10 @@ module RailsErrorDashboard
317
327
  # Job can retry it; every other failure is still swallowed.
318
328
  def initialize(exception, context = {}, worker: false)
319
329
  @exception = exception
320
- @context = context
330
+ # Invalid bytes anywhere in the context would raise as soon as it is
331
+ # JSON-encoded (ErrorContext does that in its constructor). A clean
332
+ # context costs one scan and its strings come back as the same objects.
333
+ @context = Services::EncodingSanitizer.scrub_deep(context)
321
334
  @worker = worker
322
335
  end
323
336
 
@@ -414,7 +427,7 @@ module RailsErrorDashboard
414
427
  ENV["GIT_SHA"] ||
415
428
  ENV["HEROKU_SLUG_COMMIT"] ||
416
429
  ENV["RENDER_GIT_COMMIT"] ||
417
- detect_git_sha_from_command
430
+ RailsErrorDashboard.detected_git_sha
418
431
  end
419
432
 
420
433
  if ErrorLog.column_names.include?("app_version")
@@ -434,6 +447,10 @@ module RailsErrorDashboard
434
447
  attributes[:environment] = resolve_environment
435
448
  end
436
449
 
450
+ # Neutralise invalid bytes BEFORE filtering: the filter runs regexes,
451
+ # which raise on an invalid string, and PostgreSQL rejects the INSERT.
452
+ attributes = Services::EncodingSanitizer.scrub_deep(attributes)
453
+
437
454
  # Apply sensitive data filtering (on by default)
438
455
  attributes = Services::SensitiveDataFilter.filter_attributes(attributes)
439
456
 
@@ -453,14 +470,14 @@ module RailsErrorDashboard
453
470
 
454
471
  if raw_breadcrumbs.is_a?(Array) && raw_breadcrumbs.any?
455
472
  filtered = Services::BreadcrumbCollector.filter_sensitive(raw_breadcrumbs)
456
- attributes[:breadcrumbs] = filtered.to_json
473
+ attributes[:breadcrumbs] = Services::EncodingSanitizer.scrub_deep(filtered).to_json
457
474
  end
458
475
  end
459
476
 
460
477
  # Capture system health snapshot (if enabled and column exists)
461
478
  if !storm_lite && ErrorLog.column_names.include?("system_health") && RailsErrorDashboard.configuration.enable_system_health
462
479
  health_data = @context[:_serialized_system_health] || Services::SystemHealthSnapshot.capture
463
- attributes[:system_health] = health_data.to_json
480
+ attributes[:system_health] = Services::EncodingSanitizer.scrub_deep(health_data).to_json
464
481
  end
465
482
 
466
483
  # Capture local variables (if enabled and column exists)
@@ -472,7 +489,7 @@ module RailsErrorDashboard
472
489
  raw_locals ||= @context[:_serialized_local_variables]
473
490
  if raw_locals.is_a?(Hash) && raw_locals.any?
474
491
  serialized = raw_locals == @context[:_serialized_local_variables] ? raw_locals : Services::VariableSerializer.call(raw_locals)
475
- attributes[:local_variables] = serialized.to_json
492
+ attributes[:local_variables] = Services::EncodingSanitizer.scrub_deep(serialized).to_json
476
493
  end
477
494
  rescue => e
478
495
  RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] Local variable serialization failed: #{e.message}")
@@ -496,7 +513,7 @@ module RailsErrorDashboard
496
513
  additional_filter_patterns: RailsErrorDashboard.configuration.instance_variable_filter_patterns
497
514
  )
498
515
  end
499
- attributes[:instance_variables] = serialized.to_json
516
+ attributes[:instance_variables] = Services::EncodingSanitizer.scrub_deep(serialized).to_json
500
517
  end
501
518
  rescue => e
502
519
  RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] Instance variable serialization failed: #{e.message}")
@@ -532,13 +549,16 @@ module RailsErrorDashboard
532
549
  occurred_at: attributes[:occurred_at],
533
550
  user_id: attributes[:user_id],
534
551
  request_id: error_context.request_id,
535
- session_id: error_context.session_id
552
+ # Digest, not the raw ID -- it is a bearer credential. Idempotent,
553
+ # so a value already digested at the queue boundary passes through.
554
+ session_id: Services::SensitiveDataFilter.storable_session_id(error_context.session_id)
536
555
  }
537
556
  # The release THIS event happened under. The group keeps its first
538
557
  # release; per-release counts come from here (ReleaseTimeline).
539
558
  occurrence_columns = ErrorOccurrence.column_names
540
559
  occurrence_attrs[:app_version] = attributes[:app_version] if occurrence_columns.include?("app_version")
541
560
  occurrence_attrs[:git_sha] = attributes[:git_sha] if occurrence_columns.include?("git_sha")
561
+ occurrence_attrs = Services::EncodingSanitizer.scrub_deep(occurrence_attrs)
542
562
  ErrorOccurrence.create(ErrorOccurrence.clamp_string_attributes(occurrence_attrs))
543
563
  rescue => e
544
564
  RailsErrorDashboard::Logger.error("Failed to create error occurrence: #{e.message}")
@@ -548,12 +568,12 @@ module RailsErrorDashboard
548
568
  # Send notifications for new errors and reopened errors (with throttling).
549
569
  # Muted errors skip notification dispatch but still fire plugin events.
550
570
  if error_log.occurrence_count == 1
551
- maybe_notify(error_log) { Services::NotificationThrottler.severity_meets_minimum?(error_log) }
571
+ maybe_notify(error_log, first_occurrence: true) { Services::NotificationThrottler.severity_meets_minimum?(error_log) }
552
572
  PluginRegistry.dispatch(:on_error_logged, error_log)
553
573
  trigger_callbacks(error_log)
554
574
  emit_instrumentation_events(error_log)
555
575
  elsif error_log.just_reopened
556
- maybe_notify(error_log) { Services::NotificationThrottler.should_notify?(error_log) }
576
+ maybe_notify(error_log, respect_cooldown: true) { Services::NotificationThrottler.severity_meets_minimum?(error_log) }
557
577
  PluginRegistry.dispatch(:on_error_reopened, error_log)
558
578
  trigger_callbacks(error_log)
559
579
  emit_instrumentation_events(error_log)
@@ -590,14 +610,30 @@ module RailsErrorDashboard
590
610
  # Muted errors skip notifications but still fire plugin events/callbacks.
591
611
  # During a storm (breaker not closed) per-error notifications are
592
612
  # suppressed — a single storm notification replaces them.
593
- def maybe_notify(error_log)
613
+ #
614
+ # respect_cooldown is true only for the reopened path, which is the only one
615
+ # the cooldown has ever applied to: a first occurrence and a threshold
616
+ # milestone always notify. They still stamp the row, so an error reopened
617
+ # minutes after its first notification is throttled.
618
+ def maybe_notify(error_log, respect_cooldown: false, first_occurrence: false)
594
619
  return if error_log.muted?
620
+ # wont_fix: the team has decided not to act on this error, so its
621
+ # recurrences are counted and nothing else. Plugin events still fire,
622
+ # exactly as they do for a muted error.
623
+ return if error_log.status.to_s == "wont_fix"
595
624
  return if Services::StormProtection::Gate.notifications_suppressed?
596
625
  return unless Services::NotificationThrottler.environment_allowed?(error_log)
597
626
  return unless yield
627
+ return if first_occurrence && burst_capped?
628
+
629
+ # Claim, THEN send. The claim is a conditional UPDATE only one process can
630
+ # win, so N workers reopening the same error send one notification, not
631
+ # N. The price: if the send below fails, this error is not retried inside
632
+ # the cooldown window. Recording after sending is what let every process
633
+ # through.
634
+ return unless Services::NotificationThrottler.claim!(error_log, respect_cooldown: respect_cooldown)
598
635
 
599
636
  Services::ErrorNotificationDispatcher.call(error_log)
600
- Services::NotificationThrottler.record_notification(error_log)
601
637
  rescue => e
602
638
  # The error row is already written by the time we get here. A channel
603
639
  # that cannot be reached (Redis down for the Slack job's enqueue, a
@@ -608,6 +644,37 @@ module RailsErrorDashboard
608
644
  )
609
645
  end
610
646
 
647
+ # A bad deploy can produce hundreds of DISTINCT new errors, each a first
648
+ # occurrence the per-error cooldown never sees. Past
649
+ # config.notification_burst_limit per window, new-error notifications are
650
+ # held back and ONE summary says so. Asked only for first occurrences that
651
+ # were otherwise going to notify, and before the cooldown claim, so a
652
+ # suppressed error is not stamped as notified. The error itself is already
653
+ # stored; only the notification is dropped.
654
+ def burst_capped?
655
+ # Nothing can notify, so there is nothing to cap, and no summary to enqueue.
656
+ return false unless Services::ErrorNotificationDispatcher.any_channel?
657
+
658
+ case Services::NotificationThrottler.burst_decision
659
+ when :summarize
660
+ config = RailsErrorDashboard.configuration
661
+ NotificationBurstSummaryJob.perform_later(
662
+ limit: config.notification_burst_limit.to_i,
663
+ window_seconds: config.notification_burst_window_seconds.to_i,
664
+ locale: ApplicationJob.enqueue_locale
665
+ )
666
+ true
667
+ when :suppress
668
+ true
669
+ else
670
+ false
671
+ end
672
+ rescue => e
673
+ # Fail-open: a cap that cannot decide must not cost a notification.
674
+ RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] burst cap check failed: #{e.class}: #{e.message}")
675
+ false
676
+ end
677
+
611
678
  # The environment this error is attributed to: an explicit context value
612
679
  # (truncated to the column) or the process-wide resolution. Never nil.
613
680
  def resolve_environment
@@ -704,6 +771,7 @@ module RailsErrorDashboard
704
771
  # Return early if baseline alerts are disabled or error is muted
705
772
  return unless config.enable_baseline_alerts
706
773
  return if error_log.muted?
774
+ return if error_log.status.to_s == "wont_fix" # see maybe_notify
707
775
  return unless Services::NotificationThrottler.environment_allowed?(error_log)
708
776
  return unless defined?(Queries::BaselineStats)
709
777
  return unless defined?(BaselineAlertJob)
@@ -760,15 +828,6 @@ module RailsErrorDashboard
760
828
  nil
761
829
  end
762
830
 
763
- # Detect git SHA from git command (fallback)
764
- def detect_git_sha_from_command
765
- return nil unless File.exist?(Rails.root.join(".git"))
766
- `git rev-parse --short HEAD 2>/dev/null`.strip.presence
767
- rescue => e
768
- RailsErrorDashboard::Logger.debug("Could not detect git SHA: #{e.message}")
769
- nil
770
- end
771
-
772
831
  # Detect app version from VERSION file (fallback)
773
832
  def detect_version_from_file
774
833
  version_file = Rails.root.join("VERSION")
@@ -32,6 +32,7 @@ module RailsErrorDashboard
32
32
  muted_reason: @reason
33
33
  )
34
34
 
35
+ Services::AnalyticsCacheManager.clear
35
36
  PluginRegistry.dispatch(:on_error_muted, error)
36
37
  error
37
38
  end
@@ -25,6 +25,8 @@ module RailsErrorDashboard
25
25
  resolution_reference: @resolution_data[:resolution_reference],
26
26
  status: "resolved"
27
27
  )
28
+ # The stat cards are cached; a user action must show up at once.
29
+ Services::AnalyticsCacheManager.clear
28
30
 
29
31
  # Dispatch plugin event for resolved error
30
32
  PluginRegistry.dispatch(:on_error_resolved, error)
@@ -0,0 +1,104 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RailsErrorDashboard
4
+ module Commands
5
+ # Command: repair rows that were stored with invalid bytes
6
+ #
7
+ # New captures are scrubbed on the way in (Services::EncodingSanitizer), but
8
+ # that does nothing for rows already in the table. SQLite and MySQL accept
9
+ # invalid UTF-8 and NUL bytes, and a row holding them makes the pages that
10
+ # render it answer 500. PostgreSQL rejects such rows at INSERT, so there is
11
+ # normally nothing to repair there; the scan is harmless.
12
+ #
13
+ # Behind `rake error_dashboard:scrub_invalid_encoding`. Safe to re-run: a
14
+ # second pass finds nothing to repair.
15
+ #
16
+ # @example
17
+ # ScrubInvalidEncoding.call # => { scanned: 1200, repaired: 3, unreadable: [] }
18
+ class ScrubInvalidEncoding
19
+ BATCH_SIZE = 500
20
+
21
+ MODEL_NAMES = %w[ErrorLog ErrorOccurrence].freeze
22
+
23
+ def self.call(batch_size: BATCH_SIZE)
24
+ new(batch_size: batch_size).call
25
+ end
26
+
27
+ def initialize(batch_size: BATCH_SIZE)
28
+ @batch_size = batch_size
29
+ @scanned = 0
30
+ @repaired = 0
31
+ @unreadable = []
32
+ end
33
+
34
+ # @return [Hash] { scanned:, repaired:, unreadable: [ "ErrorLog#12", ... ] }
35
+ def call
36
+ models.each { |model| scrub_model(model) }
37
+
38
+ { scanned: @scanned, repaired: @repaired, unreadable: @unreadable }
39
+ end
40
+
41
+ private
42
+
43
+ def models
44
+ MODEL_NAMES.filter_map do |name|
45
+ model = RailsErrorDashboard.const_get(name)
46
+ model if model.table_exists?
47
+ rescue => e
48
+ RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] ScrubInvalidEncoding skipped #{name}: #{e.class}")
49
+ nil
50
+ end
51
+ end
52
+
53
+ def scrub_model(model)
54
+ columns = model.columns.select { |column| %i[string text].include?(column.type) }.map(&:name)
55
+ return if columns.empty?
56
+
57
+ model.in_batches(of: @batch_size) do |batch|
58
+ ids = batch.pluck(model.primary_key)
59
+ load_rows(model, ids).each { |row| scrub_row(row, columns) }
60
+ end
61
+ end
62
+
63
+ # One query per batch; if that batch cannot be materialised at all, fall
64
+ # back to row-by-row so a single unreadable row is reported by id instead
65
+ # of hiding the 499 readable ones around it.
66
+ def load_rows(model, ids)
67
+ model.where(model.primary_key => ids).to_a
68
+ rescue => e
69
+ RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] ScrubInvalidEncoding batch read failed: #{e.class}")
70
+ ids.filter_map do |id|
71
+ model.find(id)
72
+ rescue => row_error
73
+ @unreadable << "#{model.name.demodulize}##{id} (#{row_error.class})"
74
+ nil
75
+ end
76
+ end
77
+
78
+ def scrub_row(row, columns)
79
+ @scanned += 1
80
+
81
+ changes = columns.each_with_object({}) do |column, fixed|
82
+ value = row.read_attribute(column)
83
+ next unless value.is_a?(String)
84
+
85
+ clean = Services::EncodingSanitizer.scrub(value)
86
+ fixed[column] = clean unless clean.equal?(value) || clean == value
87
+ end
88
+ # The model scrubs invalid strings as it loads a row, so by now they
89
+ # read as clean. It remembers which ones it had to repair.
90
+ row.invalid_encoding_attributes.each do |column|
91
+ changes[column] = row.read_attribute(column) if columns.include?(column)
92
+ end
93
+ return if changes.empty?
94
+
95
+ # update_columns: no callbacks, no validations, no updated_at bump. This
96
+ # is a byte-level repair, not an edit.
97
+ row.update_columns(changes)
98
+ @repaired += 1
99
+ rescue => e
100
+ @unreadable << "#{row.class.name.demodulize}##{row.id} (#{e.class})"
101
+ end
102
+ end
103
+ end
104
+ end
@@ -4,7 +4,14 @@ module RailsErrorDashboard
4
4
  module Commands
5
5
  # Command: Snooze an error for a given number of hours
6
6
  # This is a write operation that sets snoozed_until and optionally creates a comment
7
+ # Returns {success: bool, error: ErrorLog}; a failure also carries
8
+ # reason: :invalid_hours and writes nothing.
7
9
  class SnoozeError
10
+ # 30 days. The form offers at most a week; the cap is for anything that
11
+ # does not come from the form. Without one, a negative value "snoozed"
12
+ # into the past and a huge one overflowed the timestamp.
13
+ MAX_SNOOZE_HOURS = 720
14
+
8
15
  def self.call(error_id, hours:, reason: nil)
9
16
  new(error_id, hours, reason).call
10
17
  end
@@ -17,18 +24,35 @@ module RailsErrorDashboard
17
24
 
18
25
  def call
19
26
  error = ErrorLog.find(@error_id)
20
- snooze_until = @hours.hours.from_now
21
-
22
- # Store snooze reason in comments if provided
23
- if @reason.present?
24
- error.comments.create!(
25
- author_name: error.assigned_to || "System",
26
- body: "Snoozed for #{@hours} hours: #{@reason}"
27
- )
27
+ hours = whole_hours(@hours)
28
+
29
+ unless hours && (1..MAX_SNOOZE_HOURS).cover?(hours)
30
+ return { success: false, error: error, reason: :invalid_hours }
31
+ end
32
+
33
+ error.transaction do
34
+ if @reason.present?
35
+ error.comments.create!(
36
+ author_name: error.assigned_to || "System",
37
+ body: "Snoozed for #{hours} hours: #{@reason}"
38
+ )
39
+ end
40
+
41
+ error.update!(snoozed_until: hours.hours.from_now)
28
42
  end
29
43
 
30
- error.update!(snoozed_until: snooze_until)
31
- error
44
+ { success: true, error: error }
45
+ end
46
+
47
+ private
48
+
49
+ # An Integer, or a String that is one ("24"). Anything else -- a Float, a
50
+ # nested parameter, "abc" -- is nil rather than a guess.
51
+ def whole_hours(value)
52
+ case value
53
+ when Integer then value
54
+ when String then Integer(value.strip, 10, exception: false)
55
+ end
32
56
  end
33
57
  end
34
58
  end
@@ -22,6 +22,7 @@ module RailsErrorDashboard
22
22
  muted_reason: nil
23
23
  )
24
24
 
25
+ Services::AnalyticsCacheManager.clear
25
26
  PluginRegistry.dispatch(:on_error_unmuted, error)
26
27
  error
27
28
  end
@@ -4,6 +4,8 @@ module RailsErrorDashboard
4
4
  module Commands
5
5
  # Command: Update the priority level of an error
6
6
  # This is a write operation that updates the priority_level field on an ErrorLog record
7
+ # Returns {success: bool, error: ErrorLog}; a failure also carries
8
+ # reason: :invalid_priority and leaves the existing priority alone.
7
9
  class UpdateErrorPriority
8
10
  def self.call(error_id, priority_level:)
9
11
  new(error_id, priority_level).call
@@ -16,8 +18,27 @@ module RailsErrorDashboard
16
18
 
17
19
  def call
18
20
  error = ErrorLog.find(@error_id)
19
- error.update!(priority_level: @priority_level)
20
- error
21
+ level = whole_number(@priority_level)
22
+
23
+ # The column is an integer, so an unchecked "x" was cast and stored as
24
+ # 0 -- silently replacing a real priority with Low.
25
+ unless ErrorLog::PRIORITY_LEVELS.key?(level)
26
+ return { success: false, error: error, reason: :invalid_priority }
27
+ end
28
+
29
+ error.update!(priority_level: level)
30
+ { success: true, error: error }
31
+ end
32
+
33
+ private
34
+
35
+ # An Integer, or a String that is one ("3"). A nested parameter, a Float
36
+ # or free text is nil.
37
+ def whole_number(value)
38
+ case value
39
+ when Integer then value
40
+ when String then Integer(value.strip, 10, exception: false)
41
+ end
21
42
  end
22
43
  end
23
44
  end
@@ -4,7 +4,8 @@ module RailsErrorDashboard
4
4
  module Commands
5
5
  # Command: Update the status of an error with optional comment
6
6
  # This is a write operation that validates transitions and updates status
7
- # Returns {success: bool, error: ErrorLog}
7
+ # Returns {success: bool, error: ErrorLog}; a failure also carries
8
+ # reason: :unknown_status or :invalid_transition so the caller can say which.
8
9
  class UpdateErrorStatus
9
10
  def self.call(error_id, status:, comment: nil)
10
11
  new(error_id, status, comment).call
@@ -19,15 +20,17 @@ module RailsErrorDashboard
19
20
  def call
20
21
  error = ErrorLog.find(@error_id)
21
22
 
23
+ # A nested param arrives as a Hash-like object, never a known status.
24
+ unless @status.is_a?(String) && ErrorLog::STATUSES.include?(@status)
25
+ return { success: false, error: error, reason: :unknown_status }
26
+ end
27
+
22
28
  unless error.can_transition_to?(@status)
23
- return { success: false, error: error }
29
+ return { success: false, error: error, reason: :invalid_transition }
24
30
  end
25
31
 
26
32
  error.transaction do
27
- error.update!(status: @status)
28
-
29
- # Auto-resolve if status is "resolved"
30
- error.update!(resolved: true) if @status == "resolved"
33
+ error.update!(status_attributes(error))
31
34
 
32
35
  # Add comment about status change
33
36
  if @comment.present?
@@ -38,8 +41,32 @@ module RailsErrorDashboard
38
41
  end
39
42
  end
40
43
 
44
+ # The stat cards are cached; a user action must show up at once.
45
+ Services::AnalyticsCacheManager.clear
46
+
41
47
  { success: true, error: error }
42
48
  end
49
+
50
+ private
51
+
52
+ # One write, so the three columns can never disagree. resolved_at is what
53
+ # MTTR is computed from: leaving it nil (as this command used to) dropped
54
+ # every error resolved through the status workflow from the MTTR figures,
55
+ # and leaving it set on a reopened error kept a stale resolution time.
56
+ # Only "resolved" sets the flag -- wont_fix stays resolved: false.
57
+ def status_attributes(error)
58
+ attrs = { status: @status }
59
+
60
+ if @status == "resolved"
61
+ attrs[:resolved] = true
62
+ attrs[:resolved_at] = Time.current
63
+ elsif error.status == "resolved" || error.resolved?
64
+ attrs[:resolved] = false
65
+ attrs[:resolved_at] = nil
66
+ end
67
+
68
+ attrs
69
+ end
43
70
  end
44
71
  end
45
72
  end
@@ -157,6 +157,8 @@ module RailsErrorDashboard
157
157
  attr_accessor :notification_minimum_severity # Minimum severity to notify (default: :low = notify all)
158
158
  attr_accessor :notification_cooldown_minutes # Per-error cooldown in minutes (default: 5, 0 = disabled)
159
159
  attr_accessor :notification_threshold_alerts # Occurrence milestones that trigger notification (default: [10, 50, 100, 500, 1000])
160
+ attr_accessor :notification_burst_limit # Max FIRST-OCCURRENCE notifications per window, per process (default: 10, 0 = no cap)
161
+ attr_accessor :notification_burst_window_seconds # Length of that window in seconds (default: 60)
160
162
 
161
163
  # Breadcrumbs (request activity trail)
162
164
  attr_accessor :enable_breadcrumbs # Master switch (default: false)
@@ -302,7 +304,8 @@ module RailsErrorDashboard
302
304
 
303
305
  @use_separate_database = ENV.fetch("USE_SEPARATE_ERROR_DB", "false") == "true"
304
306
 
305
- # Retention policy - days to keep errors before automatic deletion (default: 90)
307
+ # Retention policy - days an error may go unseen (last_seen_at) before it is
308
+ # deleted automatically (default: 90). An error still occurring is kept.
306
309
  # Set to nil to keep errors forever (not recommended for production)
307
310
  # Schedule cleanup: RailsErrorDashboard::RetentionCleanupJob.perform_later
308
311
  @retention_days = 90
@@ -384,6 +387,8 @@ module RailsErrorDashboard
384
387
  @notification_minimum_severity = :low # Notify on all severities (current behavior)
385
388
  @notification_cooldown_minutes = 5 # 5 min cooldown per error_hash (0 = disabled)
386
389
  @notification_threshold_alerts = [ 10, 50, 100, 500, 1000 ] # Occurrence milestones
390
+ @notification_burst_limit = 10 # New-error notifications per window, per process (0 = no cap)
391
+ @notification_burst_window_seconds = 60 # One summary message replaces the rest of the window
387
392
 
388
393
  # Breadcrumbs defaults - OFF by default (opt-in)
389
394
  @enable_breadcrumbs = false # Master switch
@@ -854,6 +859,19 @@ module RailsErrorDashboard
854
859
  errors << "notification_threshold_alerts must be an Array (got: #{notification_threshold_alerts.class})"
855
860
  end
856
861
 
862
+ # Validate the first-occurrence burst cap (non-negative integers; 0 or nil
863
+ # turns the cap off)
864
+ {
865
+ notification_burst_limit: notification_burst_limit,
866
+ notification_burst_window_seconds: notification_burst_window_seconds
867
+ }.each do |name, value|
868
+ next if value.nil?
869
+
870
+ unless value.is_a?(Integer) && value >= 0
871
+ errors << "#{name} must be a non-negative Integer (got: #{value.inspect})"
872
+ end
873
+ end
874
+
857
875
  # Log warnings (non-fatal issues)
858
876
  warnings.each do |warning|
859
877
  Rails.logger.warn "[Rails Error Dashboard] #{warning}" if defined?(Rails) && Rails.respond_to?(:logger) && Rails.logger
@@ -71,6 +71,11 @@ module RailsErrorDashboard
71
71
  next
72
72
  end
73
73
 
74
+ # Resolve the running commit now (three small file reads, memoised, never
75
+ # raises) so that no capture ever pays for it. Skipped when the SHA is
76
+ # configured: then it is never consulted.
77
+ RailsErrorDashboard.detected_git_sha if RailsErrorDashboard.configuration.git_sha.blank?
78
+
74
79
  if RailsErrorDashboard.configuration.enable_error_subscriber
75
80
  Rails.error.subscribe(RailsErrorDashboard::ErrorReporter.new)
76
81
  end
@@ -188,6 +193,16 @@ module RailsErrorDashboard
188
193
  # Enable TracePoint(:raise) + TracePoint(:rescue) for swallowed exception detection
189
194
  if RailsErrorDashboard.configuration.detect_swallowed_exceptions
190
195
  RailsErrorDashboard::Services::SwallowedExceptionTracker.enable!
196
+
197
+ # Drain buffered counts at the end of every request and job. Without
198
+ # this the buffer is only ever drained by a LATER rescue on the SAME
199
+ # thread, so a swallowed exception that happens once stays invisible
200
+ # until the process exits. to_complete fires after the response body
201
+ # is closed, so it never delays a request (safety rule 2); the flush is
202
+ # deadline-gated, so a flood is still one write per interval.
203
+ Rails.application.executor.to_complete do
204
+ RailsErrorDashboard::Services::SwallowedExceptionTracker.flush_if_due!
205
+ end
191
206
  end
192
207
 
193
208
  # Import crash files from previous process death, then register at_exit hook
@@ -17,7 +17,7 @@ module RailsErrorDashboard
17
17
 
18
18
  def call
19
19
  # Cache analytics data for 5 minutes to reduce database load
20
- # Cache key includes days parameter and last error update timestamp
20
+ # Cache key includes the days parameter and the cache generation
21
21
  Rails.cache.fetch(cache_key, expires_in: 5.minutes) do
22
22
  {
23
23
  days: @days,
@@ -41,13 +41,14 @@ module RailsErrorDashboard
41
41
  # - Query class name
42
42
  # - Days parameter (different time ranges = different caches)
43
43
  # - Application ID (per-app caching)
44
- # - Last error update timestamp (auto-invalidates when errors change)
44
+ # - The cache generation (bumped by user actions; see AnalyticsCacheManager).
45
+ # Captures do not bump it: they rely on the 5-minute TTL.
45
46
  # - Start date (ensures correct time window)
46
47
  [
47
48
  "analytics_stats",
48
49
  @days,
49
50
  @application_id || "all",
50
- base_scope.maximum(:updated_at)&.to_i || 0,
51
+ Services::AnalyticsCacheManager.generation,
51
52
  @start_date.to_date.to_s
52
53
  ].join("/")
53
54
  end