rails_error_dashboard 0.12.1 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. checksums.yaml +4 -4
  2. data/app/controllers/rails_error_dashboard/application_controller.rb +92 -11
  3. data/app/controllers/rails_error_dashboard/errors_controller.rb +111 -34
  4. data/app/controllers/rails_error_dashboard/webhooks_controller.rb +3 -0
  5. data/app/helpers/rails_error_dashboard/application_helper.rb +46 -1
  6. data/app/helpers/rails_error_dashboard/backtrace_helper.rb +8 -0
  7. data/app/jobs/rails_error_dashboard/async_error_logging_job.rb +10 -0
  8. data/app/jobs/rails_error_dashboard/concerns/plain_channel_message.rb +71 -0
  9. data/app/jobs/rails_error_dashboard/notification_burst_summary_job.rb +55 -0
  10. data/app/jobs/rails_error_dashboard/retention_cleanup_job.rb +122 -1
  11. data/app/jobs/rails_error_dashboard/storm_flush_job.rb +7 -4
  12. data/app/jobs/rails_error_dashboard/storm_notification_job.rb +8 -44
  13. data/app/models/rails_error_dashboard/error_baseline.rb +12 -9
  14. data/app/models/rails_error_dashboard/error_comment.rb +0 -5
  15. data/app/models/rails_error_dashboard/error_log.rb +19 -3
  16. data/app/models/rails_error_dashboard/error_logs_record.rb +34 -0
  17. data/app/models/rails_error_dashboard/error_occurrence.rb +9 -1
  18. data/app/models/rails_error_dashboard/event_count.rb +132 -0
  19. data/app/models/rails_error_dashboard/event_timing_gap.rb +55 -0
  20. data/app/views/layouts/rails_error_dashboard.html.erb +58 -5
  21. data/app/views/rails_error_dashboard/errors/_discussion.html.erb +18 -11
  22. data/app/views/rails_error_dashboard/errors/_issue_section.html.erb +22 -8
  23. data/app/views/rails_error_dashboard/errors/_request_context.html.erb +2 -0
  24. data/app/views/rails_error_dashboard/errors/_sidebar_metadata.html.erb +15 -6
  25. data/app/views/rails_error_dashboard/errors/analytics.html.erb +8 -8
  26. data/app/views/rails_error_dashboard/errors/diagnostic_dumps.html.erb +1 -1
  27. data/app/views/rails_error_dashboard/errors/index.html.erb +6 -1
  28. data/app/views/rails_error_dashboard/errors/overview.html.erb +12 -0
  29. data/app/views/rails_error_dashboard/errors/platform_comparison.html.erb +7 -7
  30. data/app/views/rails_error_dashboard/errors/releases.html.erb +2 -2
  31. data/app/views/rails_error_dashboard/errors/settings.html.erb +2 -0
  32. data/app/views/rails_error_dashboard/errors/show.html.erb +3 -3
  33. data/config/locales/de.yml +30 -0
  34. data/config/locales/en.yml +37 -0
  35. data/config/locales/es.yml +30 -0
  36. data/config/locales/fr.yml +30 -0
  37. data/config/locales/it.yml +30 -0
  38. data/config/locales/ja.yml +30 -0
  39. data/config/locales/pl.yml +30 -0
  40. data/config/locales/pt-BR.yml +30 -0
  41. data/config/locales/ru.yml +30 -0
  42. data/config/locales/uk.yml +30 -0
  43. data/config/locales/zh-CN.yml +30 -0
  44. data/db/migrate/20260917000001_add_last_notified_at_to_error_logs.rb +40 -0
  45. data/db/migrate/20260919000001_create_event_counts.rb +71 -0
  46. data/db/migrate/20260920000001_add_buckets_incomplete_to_storm_events.rb +25 -0
  47. data/db/migrate/20260920000002_create_event_timing_gaps.rb +55 -0
  48. data/lib/generators/rails_error_dashboard/install/templates/initializer.rb +12 -1
  49. data/lib/rails_error_dashboard/commands/assign_error.rb +12 -2
  50. data/lib/rails_error_dashboard/commands/backfill_environments.rb +2 -0
  51. data/lib/rails_error_dashboard/commands/backfill_resolved_at.rb +42 -0
  52. data/lib/rails_error_dashboard/commands/batch_delete_errors.rb +1 -0
  53. data/lib/rails_error_dashboard/commands/batch_mute_errors.rb +2 -0
  54. data/lib/rails_error_dashboard/commands/batch_resolve_errors.rb +2 -0
  55. data/lib/rails_error_dashboard/commands/batch_unmute_errors.rb +2 -0
  56. data/lib/rails_error_dashboard/commands/find_or_increment_error.rb +109 -11
  57. data/lib/rails_error_dashboard/commands/flush_rack_attack_events.rb +3 -1
  58. data/lib/rails_error_dashboard/commands/flush_storm_counts.rb +252 -12
  59. data/lib/rails_error_dashboard/commands/flush_swallowed_exceptions.rb +5 -2
  60. data/lib/rails_error_dashboard/commands/link_existing_issue.rb +1 -0
  61. data/lib/rails_error_dashboard/commands/log_error.rb +291 -40
  62. data/lib/rails_error_dashboard/commands/mute_error.rb +1 -0
  63. data/lib/rails_error_dashboard/commands/resolve_error.rb +2 -0
  64. data/lib/rails_error_dashboard/commands/scrub_invalid_encoding.rb +104 -0
  65. data/lib/rails_error_dashboard/commands/snooze_error.rb +34 -10
  66. data/lib/rails_error_dashboard/commands/unmute_error.rb +1 -0
  67. data/lib/rails_error_dashboard/commands/update_error_priority.rb +23 -2
  68. data/lib/rails_error_dashboard/commands/update_error_status.rb +33 -6
  69. data/lib/rails_error_dashboard/configuration.rb +39 -1
  70. data/lib/rails_error_dashboard/engine.rb +28 -0
  71. data/lib/rails_error_dashboard/manual_error_reporter.rb +16 -5
  72. data/lib/rails_error_dashboard/queries/analytics_stats.rb +89 -30
  73. data/lib/rails_error_dashboard/queries/baseline_stats.rb +107 -0
  74. data/lib/rails_error_dashboard/queries/dashboard_stats.rb +216 -74
  75. data/lib/rails_error_dashboard/queries/error_correlation.rb +6 -3
  76. data/lib/rails_error_dashboard/queries/errors_list.rb +15 -2
  77. data/lib/rails_error_dashboard/queries/event_volume.rb +503 -0
  78. data/lib/rails_error_dashboard/queries/similar_errors.rb +1 -1
  79. data/lib/rails_error_dashboard/services/analytics_cache_manager.rb +43 -18
  80. data/lib/rails_error_dashboard/services/backtrace_processor.rb +3 -1
  81. data/lib/rails_error_dashboard/services/breadcrumb_collector.rb +23 -0
  82. data/lib/rails_error_dashboard/services/cause_chain_extractor.rb +3 -1
  83. data/lib/rails_error_dashboard/services/codeberg_issue_client.rb +13 -2
  84. data/lib/rails_error_dashboard/services/diagnostic_dump_generator.rb +5 -3
  85. data/lib/rails_error_dashboard/services/encoding_sanitizer.rb +80 -0
  86. data/lib/rails_error_dashboard/services/error_broadcaster.rb +180 -32
  87. data/lib/rails_error_dashboard/services/error_hash_generator.rb +9 -3
  88. data/lib/rails_error_dashboard/services/error_notification_dispatcher.rb +13 -0
  89. data/lib/rails_error_dashboard/services/exception_filter.rb +55 -0
  90. data/lib/rails_error_dashboard/services/git_head_reader.rb +102 -0
  91. data/lib/rails_error_dashboard/services/notification_throttler.rb +172 -28
  92. data/lib/rails_error_dashboard/services/sensitive_data_filter.rb +47 -1
  93. data/lib/rails_error_dashboard/services/storm_protection/circuit_breaker.rb +59 -3
  94. data/lib/rails_error_dashboard/services/storm_protection/count_buffer.rb +64 -6
  95. data/lib/rails_error_dashboard/services/storm_protection/fingerprint_buckets.rb +25 -2
  96. data/lib/rails_error_dashboard/services/storm_protection/gate.rb +32 -5
  97. data/lib/rails_error_dashboard/services/swallowed_exception_tracker.rb +99 -23
  98. data/lib/rails_error_dashboard/services/url_safety.rb +40 -0
  99. data/lib/rails_error_dashboard/services/variable_serializer.rb +125 -10
  100. data/lib/rails_error_dashboard/subscribers/breadcrumb_subscriber.rb +111 -0
  101. data/lib/rails_error_dashboard/subscribers/issue_tracker_subscriber.rb +11 -2
  102. data/lib/rails_error_dashboard/value_objects/error_context.rb +40 -3
  103. data/lib/rails_error_dashboard/version.rb +1 -1
  104. data/lib/rails_error_dashboard.rb +34 -0
  105. data/lib/tasks/error_dashboard.rake +54 -4
  106. metadata +16 -2
@@ -441,6 +441,7 @@ ru:
441
441
  issue_section:
442
442
  title: Трекер задач
443
443
  view_issue: Открыть задачу
444
+ unsafe_link: "Эта ссылка на задачу не является URL-адресом http(s), поэтому она показана как текст и не может быть открыта."
444
445
  linked: Привязана
445
446
  issue_reference: "%{provider} #%{number}"
446
447
  states:
@@ -613,6 +614,7 @@ ru:
613
614
  snapshot_stale: из более раннего появления — последующие не содержали контекста
614
615
  snapshot_fidelity_lite: сокращённый сбор (защита от storm отбросила payload контекста)
615
616
  snapshot_fidelity_minimal: учтено во время storm — контекст не собирался
617
+ snapshot_fidelity_partial: смешанный снимок — часть полей осталась от более раннего события
616
618
  snapshot_unknown: получен до того, как RED начал записывать происхождение снимков
617
619
  similar_errors:
618
620
  title: Похожие ошибки
@@ -980,6 +982,7 @@ ru:
980
982
  rate_per_hour: "%{value}/ч"
981
983
  affected_users_incomplete: не менее — защита от storm отбросила детализацию по событиям
982
984
  data_unavailable: Статистика сейчас недоступна — приведённые ниже числа не являются измерением.
985
+ event_timing_incomplete: "Отметки времени неполные — защита от шторма отбросила их; итоговые значения остаются точными."
983
986
  unresolved: Не решено
984
987
  unresolved_hint: Ожидают решения
985
988
  resolution_rate: Доля решённых
@@ -1019,6 +1022,16 @@ ru:
1019
1022
  other: "%{count} типа"
1020
1023
  few: "%{count} типа"
1021
1024
  many: "%{count} типов"
1025
+ errors_page:
1026
+ csrf:
1027
+ title: "Срок действия токена сеанса истёк"
1028
+ message: "Вернитесь назад, перезагрузите страницу и повторите попытку. Ничего не было изменено."
1029
+ bad_request:
1030
+ title: "Не удалось обработать этот запрос"
1031
+ message: "Обязательное значение отсутствовало или имело неверный формат. Вернитесь назад и повторите попытку."
1032
+ not_acceptable:
1033
+ title: "Этот формат недоступен"
1034
+ message: "Панель не может сформировать формат, запрошенный в этом запросе."
1022
1035
  settings:
1023
1036
  title: Настройки
1024
1037
  version: v%{version}
@@ -1151,6 +1164,8 @@ ru:
1151
1164
  application_name: Имя этого приложения (определяется автоматически, если не
1152
1165
  задано)
1153
1166
  environment: Окружение, к которому относятся ошибки (Rails.env, если не задано)
1167
+ notification_burst_limit: "Уведомления о новых ошибках за интервал на процесс (0 = без ограничения)"
1168
+ notification_burst_window_seconds: "Длительность интервала уведомлений о новых ошибках"
1154
1169
  notification_environments: Уведомлять только для этих окружений (все, если
1155
1170
  не задано)
1156
1171
  database: Активная база данных для журнала ошибок
@@ -1261,6 +1276,16 @@ ru:
1261
1276
  linked: Задача успешно привязана
1262
1277
  create_failed: 'Не удалось создать задачу: %{reason}'
1263
1278
  link_failed: 'Не удалось привязать задачу: %{reason}'
1279
+ status:
1280
+ updated: "Статус изменён на %{status}"
1281
+ invalid_transition: "Нельзя изменить статус с %{from} на %{to}"
1282
+ unknown: "Такого статуса не существует"
1283
+ snooze:
1284
+ invalid_hours: "Время откладывания должно быть целым числом часов от 1 до %{max}"
1285
+ priority:
1286
+ invalid: "Такого уровня приоритета не существует"
1287
+ assign:
1288
+ blank: "Введите имя того, кому назначить эту ошибку"
1264
1289
  not_enabled: "%{feature} не включено. Включите в %{file}"
1265
1290
  not_enabled_plural: "%{feature} не включены. Включите их в %{file}"
1266
1291
  not_enabled_options: "%{feature} не включено. Включите %{options} в %{file}"
@@ -1303,6 +1328,7 @@ ru:
1303
1328
  not_configured: Отслеживание задач не настроено
1304
1329
  already_linked: 'К ошибке уже привязана задача: %{url}'
1305
1330
  url_required: Требуется URL задачи
1331
+ url_invalid: URL задачи должен начинаться с http:// или https://
1306
1332
  error_not_found: 'Ошибка не найдена: %{id}'
1307
1333
  batch_failed:
1308
1334
  mute:
@@ -1441,6 +1467,10 @@ ru:
1441
1467
  few: "%{value} ошибки"
1442
1468
  many: "%{value} ошибок"
1443
1469
  view_in_dashboard: Открыть в панели
1470
+ burst:
1471
+ suppressed: "В приложении %{application} за %{window} секунд зафиксировано более %{limit} новых ошибок."
1472
+ still_recorded: "Уведомления о дальнейших новых ошибках приостановлены до конца этого интервала; каждая ошибка по-прежнему записывается."
1473
+ dashboard: "Панель: %{url}"
1444
1474
  storm:
1445
1475
  detected: Обнаружен шторм ошибок в %{application} в %{started_at}.
1446
1476
  mode:
@@ -438,6 +438,7 @@ uk:
438
438
  issue_section:
439
439
  title: Трекер задач
440
440
  view_issue: Переглянути задачу
441
+ unsafe_link: "Це посилання на задачу не є URL-адресою http(s), тому воно показане як текст і не може бути відкрите."
441
442
  linked: Приєднано
442
443
  issue_reference: "%{provider} #%{number}"
443
444
  states:
@@ -611,6 +612,7 @@ uk:
611
612
  snapshot_stale: з ранішого випадку — подальші не містили контексту
612
613
  snapshot_fidelity_lite: скорочений збір (захист від storm відкинув payload контексту)
613
614
  snapshot_fidelity_minimal: враховано під час storm — контекст не збирався
615
+ snapshot_fidelity_partial: змішаний знімок — частина полів залишилася від давнішої події
614
616
  snapshot_unknown: отримано до того, як RED почав записувати походження знімків
615
617
  similar_errors:
616
618
  title: Схожі помилки
@@ -977,6 +979,7 @@ uk:
977
979
  rate_per_hour: "%{value}/год"
978
980
  affected_users_incomplete: щонайменше — захист від storm відкинув деталізацію за подіями
979
981
  data_unavailable: Статистика зараз недоступна — наведені нижче числа не є вимірюванням.
982
+ event_timing_incomplete: "Позначки часу неповні — захист від шторму відкинув їх; підсумкові значення залишаються точними."
980
983
  unresolved: Не вирішено
981
984
  unresolved_hint: Очікують вирішення
982
985
  resolution_rate: Частка вирішених
@@ -1016,6 +1019,16 @@ uk:
1016
1019
  other: "%{count} типу"
1017
1020
  few: "%{count} типи"
1018
1021
  many: "%{count} типів"
1022
+ errors_page:
1023
+ csrf:
1024
+ title: "Термін дії токена сеансу минув"
1025
+ message: "Поверніться назад, перезавантажте сторінку та спробуйте ще раз. Нічого не було змінено."
1026
+ bad_request:
1027
+ title: "Не вдалося обробити цей запит"
1028
+ message: "Обов'язкове значення було відсутнє або мало неправильний формат. Поверніться назад і спробуйте ще раз."
1029
+ not_acceptable:
1030
+ title: "Цей формат недоступний"
1031
+ message: "Панель не може сформувати формат, запитаний у цьому запиті."
1019
1032
  settings:
1020
1033
  title: Налаштування
1021
1034
  version: v%{version}
@@ -1147,6 +1160,8 @@ uk:
1147
1160
  application_name: Назва цього застосунку (визначається автоматично, якщо не
1148
1161
  задано)
1149
1162
  environment: Середовище, до якого належать помилки (Rails.env, якщо не задано)
1163
+ notification_burst_limit: "Сповіщення про нові помилки за інтервал на процес (0 = без обмеження)"
1164
+ notification_burst_window_seconds: "Тривалість інтервалу сповіщень про нові помилки"
1150
1165
  notification_environments: Сповіщати лише для цих середовищ (усі, якщо не
1151
1166
  задано)
1152
1167
  database: Активна база даних для журналів помилок
@@ -1260,6 +1275,16 @@ uk:
1260
1275
  linked: Задачу успішно приєднано
1261
1276
  create_failed: 'Не вдалося створити задачу: %{reason}'
1262
1277
  link_failed: 'Не вдалося приєднати задачу: %{reason}'
1278
+ status:
1279
+ updated: "Статус змінено на %{status}"
1280
+ invalid_transition: "Неможливо змінити статус з %{from} на %{to}"
1281
+ unknown: "Такого статусу не існує"
1282
+ snooze:
1283
+ invalid_hours: "Час відкладення має бути цілим числом годин від 1 до %{max}"
1284
+ priority:
1285
+ invalid: "Такого рівня пріоритету не існує"
1286
+ assign:
1287
+ blank: "Введіть ім'я того, кому призначити цю помилку"
1263
1288
  not_enabled: "%{feature} не увімкнено. Увімкніть у %{file}"
1264
1289
  not_enabled_plural: "%{feature} не увімкнено. Увімкніть їх у %{file}"
1265
1290
  not_enabled_options: "%{feature} не увімкнено. Увімкніть %{options} у %{file}"
@@ -1301,6 +1326,7 @@ uk:
1301
1326
  not_configured: Відстеження задач не налаштовано
1302
1327
  already_linked: 'До помилки вже приєднано задачу: %{url}'
1303
1328
  url_required: Потрібна URL-адреса задачі
1329
+ url_invalid: URL-адреса задачі має починатися з http:// або https://
1304
1330
  error_not_found: 'Помилку не знайдено: %{id}'
1305
1331
  batch_failed:
1306
1332
  mute:
@@ -1439,6 +1465,10 @@ uk:
1439
1465
  few: "%{value} помилки"
1440
1466
  many: "%{value} помилок"
1441
1467
  view_in_dashboard: Переглянути в панелі
1468
+ burst:
1469
+ suppressed: "У застосунку %{application} за %{window} секунд зафіксовано понад %{limit} нових помилок."
1470
+ still_recorded: "Сповіщення про подальші нові помилки призупинено до кінця цього інтервалу; кожна помилка, як і раніше, записується."
1471
+ dashboard: "Панель: %{url}"
1442
1472
  storm:
1443
1473
  detected: Виявлено шторм помилок у %{application} о %{started_at}.
1444
1474
  mode:
@@ -380,6 +380,7 @@ zh-CN:
380
380
  issue_section:
381
381
  title: 工单系统
382
382
  view_issue: 查看工单
383
+ unsafe_link: "此工单链接不是 http(s) URL,因此仅以文本显示,无法打开。"
383
384
  linked: 已关联
384
385
  issue_reference: "%{provider} #%{number}"
385
386
  states:
@@ -528,6 +529,7 @@ zh-CN:
528
529
  snapshot_stale: 来自更早的一次发生 — 之后的发生未携带上下文
529
530
  snapshot_fidelity_lite: 已精简的捕获(storm 保护丢弃了上下文 payload)
530
531
  snapshot_fidelity_minimal: 在 storm 期间计数 — 未捕获任何上下文
532
+ snapshot_fidelity_partial: 混合快照 — 部分字段来自更早的一次发生
531
533
  snapshot_unknown: 在 RED 记录快照来源之前捕获
532
534
  similar_errors:
533
535
  title: 相似错误
@@ -841,6 +843,7 @@ zh-CN:
841
843
  rate_per_hour: "%{value}/小时"
842
844
  affected_users_incomplete: 至少 — storm 保护丢弃了逐事件的明细
843
845
  data_unavailable: 统计数据当前不可用 — 下方数字并非测量结果。
846
+ event_timing_incomplete: "事件时间记录不完整 — 风暴保护丢弃了时间戳;总数仍然准确。"
844
847
  unresolved: 未解决
845
848
  unresolved_hint: 等待处理
846
849
  resolution_rate: 解决率
@@ -870,6 +873,16 @@ zh-CN:
870
873
  user_types_html: 用户 %{user} · %{types}
871
874
  types:
872
875
  other: "%{count} 种类型"
876
+ errors_page:
877
+ csrf:
878
+ title: "会话令牌已过期"
879
+ message: "请返回并重新加载页面后重试。未做任何更改。"
880
+ bad_request:
881
+ title: "无法理解此请求"
882
+ message: "缺少必需的值或其格式不正确。请返回后重试。"
883
+ not_acceptable:
884
+ title: "此格式不可用"
885
+ message: "仪表板无法生成此请求所要求的格式。"
873
886
  settings:
874
887
  title: 设置
875
888
  version: v%{version}
@@ -963,6 +976,8 @@ zh-CN:
963
976
  sampling_rate: 错误捕获的采样比例
964
977
  application_name: 本应用的名称(未设置时自动检测)
965
978
  environment: 错误所属的环境(未设置时为 Rails.env)
979
+ notification_burst_limit: "每个时间窗口、每个进程的新错误通知数(0 = 不限制)"
980
+ notification_burst_window_seconds: "新错误通知时间窗口的长度"
966
981
  notification_environments: 仅对这些环境发送通知(未设置时为全部)
967
982
  database: 错误日志使用的数据库
968
983
  use_separate_database: 为错误日志使用独立数据库
@@ -1049,6 +1064,16 @@ zh-CN:
1049
1064
  linked: 工单关联成功
1050
1065
  create_failed: 创建工单失败:%{reason}
1051
1066
  link_failed: 关联工单失败:%{reason}
1067
+ status:
1068
+ updated: "状态已更改为 %{status}"
1069
+ invalid_transition: "无法将状态从 %{from} 更改为 %{to}"
1070
+ unknown: "该状态不存在"
1071
+ snooze:
1072
+ invalid_hours: "延后时间必须是 1 到 %{max} 之间的整数小时"
1073
+ priority:
1074
+ invalid: "该优先级不存在"
1075
+ assign:
1076
+ blank: "请输入要分配此错误的人员姓名"
1052
1077
  not_enabled: "%{feature}未启用。请在 %{file} 中启用。"
1053
1078
  not_enabled_plural: "%{feature}未启用。请在 %{file} 中启用。"
1054
1079
  not_enabled_options: "%{feature}未启用。请在 %{file} 中启用 %{options}。"
@@ -1087,6 +1112,7 @@ zh-CN:
1087
1112
  not_configured: 工单跟踪尚未配置
1088
1113
  already_linked: 该错误已关联工单:%{url}
1089
1114
  url_required: 必须提供工单 URL
1115
+ url_invalid: 工单 URL 必须以 http:// 或 https:// 开头
1090
1116
  error_not_found: 未找到错误:%{id}
1091
1117
  batch_failed:
1092
1118
  mute:
@@ -1203,6 +1229,10 @@ zh-CN:
1203
1229
  threshold_errors:
1204
1230
  other: "%{value} 个错误"
1205
1231
  view_in_dashboard: 在面板中查看
1232
+ burst:
1233
+ suppressed: "%{application} 在 %{window} 秒内报告了超过 %{limit} 个新错误。"
1234
+ still_recorded: "在该时间窗口结束之前,后续新错误的通知将暂停;所有错误仍会被记录。"
1235
+ dashboard: "面板:%{url}"
1206
1236
  storm:
1207
1237
  detected: 在 %{application} 于 %{started_at} 检测到错误风暴。
1208
1238
  mode:
@@ -0,0 +1,40 @@
1
+ # frozen_string_literal: true
2
+
3
+ # When a notification was last sent for this error group.
4
+ #
5
+ # The notification cooldown used to live in a Hash inside each process. Every
6
+ # Puma worker and every job process had its own copy, so one bad deploy that
7
+ # reopened an error notified once PER PROCESS, and a restart forgot the
8
+ # cooldown entirely.
9
+ #
10
+ # With the timestamp on the row the cooldown is claimed in the database with a
11
+ # single conditional UPDATE, which exactly one process can win:
12
+ #
13
+ # UPDATE ... SET last_notified_at = now
14
+ # WHERE id = ? AND (last_notified_at IS NULL OR last_notified_at < cutoff)
15
+ #
16
+ # No index: the row is always addressed by primary key.
17
+ #
18
+ # Nullable, no default: NULL means "never notified". Until this migration has
19
+ # run the gem falls back to the in-process cooldown, so upgrading the gem
20
+ # before migrating is safe.
21
+ #
22
+ # up/down rather than `change`: the column_exists? guard that makes `up` safe
23
+ # to re-run would make a reversed `change` skip the removal.
24
+ class AddLastNotifiedAtToErrorLogs < ActiveRecord::Migration[7.0]
25
+ TABLE = :rails_error_dashboard_error_logs
26
+
27
+ def up
28
+ return unless table_exists?(TABLE)
29
+ return if column_exists?(TABLE, :last_notified_at)
30
+
31
+ add_column TABLE, :last_notified_at, :datetime
32
+ end
33
+
34
+ def down
35
+ return unless table_exists?(TABLE)
36
+ return unless column_exists?(TABLE, :last_notified_at)
37
+
38
+ remove_column TABLE, :last_notified_at
39
+ end
40
+ end
@@ -0,0 +1,71 @@
1
+ # frozen_string_literal: true
2
+
3
+ # Time buckets for storm-shed event volume.
4
+ #
5
+ # Every time-window figure on the dashboard ("errors today", the daily trend,
6
+ # the hourly chart, the error rate) used to filter error_logs by occurred_at
7
+ # and then SUM(occurrence_count). occurred_at is FIRST-SEEN and is never
8
+ # rewritten on recurrence, so that sum reports the lifetime volume of the
9
+ # groups born inside the window -- not the events that happened in it. An
10
+ # error first seen at 23:59 and recurring at 00:01 reported "0 errors today"
11
+ # and put both events on yesterday.
12
+ #
13
+ # The obvious fix -- count error_occurrences rows in the window -- is wrong on
14
+ # its own. Storm protection deliberately writes NO occurrence row while
15
+ # shedding: it folds N events into occurrence_count and nothing else. Counting
16
+ # occurrence rows would therefore erase storm volume from every time window,
17
+ # which is exactly when the numbers matter most.
18
+ #
19
+ # So shed volume needs its own timestamp, and this is it: one row per
20
+ # (error group, 15-minute bucket), holding how many shed events landed in it.
21
+ # Window volume is then
22
+ #
23
+ # occurrence rows in the window + shed bucket counts in the window
24
+ #
25
+ # which is correct for both ordinary and shed events.
26
+ #
27
+ # Small by construction: a row is written only while a storm is actually
28
+ # shedding, at most one per group per quarter hour.
29
+ #
30
+ # Cleanup follows the GROUP, not the bucket's own age: RetentionCleanupJob
31
+ # deletes the buckets of groups it expires, and ErrorLog has_many :event_counts
32
+ # with dependent: :delete_all covers an explicit destroy. Buckets of a
33
+ # still-active group are deliberately kept -- pruning them by bucket_at alone
34
+ # would silently redistribute those events onto the group's first-seen date,
35
+ # because EventVolume falls back to the group's lifetime count for any group
36
+ # with no per-event record left.
37
+ #
38
+ # (An earlier version of this comment claimed pruning "on bucket_at" that was
39
+ # never implemented. Both mechanisms above are now covered by
40
+ # spec/models/rails_error_dashboard/event_count_cleanup_spec.rb.)
41
+ class CreateEventCounts < ActiveRecord::Migration[7.0]
42
+ def change
43
+ # Guard against a squashed schema migration having already created this
44
+ # table -- without it, every later migration is silently cancelled.
45
+ return if table_exists?(:rails_error_dashboard_event_counts)
46
+
47
+ create_table :rails_error_dashboard_event_counts do |t|
48
+ t.bigint :error_log_id, null: false
49
+ # Truncated to a 15-MINUTE bucket, in UTC (EventCount::BUCKET_SECONDS,
50
+ # shared with the producer). Not the hour: every UTC offset in use
51
+ # divides into 15 minutes -- including +05:30 and +05:45 -- so a local
52
+ # midnight falls on a bucket EDGE and a day's total is exact. The row
53
+ # count stays bounded even through a long storm: at most one row per
54
+ # group per quarter hour, and only while shedding.
55
+ t.datetime :bucket_at, null: false
56
+ t.bigint :count, null: false, default: 0
57
+ t.timestamps
58
+ end
59
+
60
+ # The upsert target: one bucket per group per quarter hour. Named explicitly --
61
+ # an auto-generated name here would exceed PostgreSQL's 63-character limit
62
+ # and fail during the HOST app's deploy (see mailboxer#480).
63
+ add_index :rails_error_dashboard_event_counts, [ :error_log_id, :bucket_at ],
64
+ unique: true,
65
+ name: "index_red_event_counts_on_group_and_bucket"
66
+
67
+ # Window queries scan by time; retention prunes on the same column.
68
+ add_index :rails_error_dashboard_event_counts, :bucket_at,
69
+ name: "index_red_event_counts_on_bucket_at"
70
+ end
71
+ end
@@ -0,0 +1,25 @@
1
+ # frozen_string_literal: true
2
+
3
+ # Records that a storm episode lost its per-bucket TIMING evidence.
4
+ #
5
+ # When the rollup table is permanently unavailable (never migrated, or an
6
+ # adapter that refuses the statement), FlushStormCounts deliberately degrades:
7
+ # it keeps the authoritative lifetime occurrence_count and skips the bucket.
8
+ # That is the right trade -- losing a time bucket beats losing a count -- but
9
+ # it leaves every time-window figure for that episode resting on the group's
10
+ # own occurred_at rather than on per-event evidence.
11
+ #
12
+ # The flag was returned in the flush result and then dropped on the floor: no
13
+ # caller persisted it, no query read it, and nothing on the dashboard said so.
14
+ # A replay lost it entirely. Persisting it on the episode is what lets the
15
+ # Overview report an incomplete dimension instead of presenting a figure of
16
+ # unknown completeness as fact -- exactly as affected_users_incomplete already
17
+ # does for storm-shed occurrence rows.
18
+ class AddBucketsIncompleteToStormEvents < ActiveRecord::Migration[7.0]
19
+ def change
20
+ return unless table_exists?(:rails_error_dashboard_storm_events)
21
+ return if column_exists?(:rails_error_dashboard_storm_events, :buckets_incomplete)
22
+
23
+ add_column :rails_error_dashboard_storm_events, :buckets_incomplete, :boolean, default: false
24
+ end
25
+ end
@@ -0,0 +1,55 @@
1
+ # frozen_string_literal: true
2
+
3
+ # One row per interval whose per-event TIMING evidence was lost.
4
+ #
5
+ # When the rollup table is permanently unavailable, FlushStormCounts keeps the
6
+ # authoritative lifetime occurrence_count and skips the bucket. The counts stay
7
+ # exact; only their placement in time becomes approximate. The dashboard has to
8
+ # say so, or it presents a figure of unknown completeness as fact.
9
+ #
10
+ # Three earlier attempts to carry that fact failed, each for a structural
11
+ # reason, and this table exists to fix all three at once:
12
+ #
13
+ # 1. A boolean in the command's RETURN HASH. A background job's return value
14
+ # never reaches the dashboard.
15
+ # 2. A flag on the storm EPISODE. The episode is OPTIONAL -- the gate can
16
+ # shed events with its breaker closed and pass episode: nil -- so the
17
+ # warning silently vanished exactly when no episode existed.
18
+ # 3. Written by upsert_storm_event, which runs AFTER the counts transaction
19
+ # commits and rescues its own failures. A transient save failure lost the
20
+ # marker while the batch ledger recorded the batch as applied, so the
21
+ # replay was suppressed and the gap was never recorded.
22
+ #
23
+ # So: its own table, written INSIDE the same transaction as the counts (either
24
+ # both land or neither does), keyed by the interval it describes rather than by
25
+ # an episode that may not exist.
26
+ #
27
+ # Bounded by construction: one row per (application, bucket) per degraded
28
+ # flush, and only while the rollup is actually unusable -- which is a
29
+ # misconfiguration, not a steady state. Retention prunes it by covered_until.
30
+ class CreateEventTimingGaps < ActiveRecord::Migration[7.0]
31
+ def change
32
+ return if table_exists?(:rails_error_dashboard_event_timing_gaps)
33
+
34
+ create_table :rails_error_dashboard_event_timing_gaps do |t|
35
+ # Nullable: a gap can predate application scoping, and a NULL here means
36
+ # "applies to every application" rather than "unknown".
37
+ t.bigint :application_id
38
+ # The interval whose timing is unreliable. covered_from is the earliest
39
+ # event in the degraded flush, covered_until the latest.
40
+ t.datetime :covered_from, null: false
41
+ t.datetime :covered_until, null: false
42
+ # How many events lost their timestamps, so the UI can say how much of
43
+ # the window is affected rather than only that something is.
44
+ t.bigint :events_affected, null: false, default: 0
45
+ t.timestamps
46
+ end
47
+
48
+ # The read is "does any gap overlap the window I am displaying", which
49
+ # scans on covered_until. Named explicitly -- an auto-generated name here
50
+ # would exceed PostgreSQL's 63-character limit and fail during the HOST
51
+ # app's deploy (see mailboxer#480).
52
+ add_index :rails_error_dashboard_event_timing_gaps, :covered_until,
53
+ name: "index_red_timing_gaps_on_covered_until"
54
+ end
55
+ end
@@ -21,6 +21,16 @@ RailsErrorDashboard.configure do |config|
21
21
  # Only notify (Slack, email, Discord, PagerDuty, webhooks, storm and
22
22
  # baseline alerts) for these environments. nil = every environment.
23
23
  # config.notification_environments = %w[production]
24
+ #
25
+ # Notification throttling. The per-error cooldown stops one error that keeps
26
+ # being reopened from paging repeatedly; it is held in the database, so it
27
+ # applies across every worker. The burst cap is for the deploy that throws
28
+ # hundreds of DIFFERENT new errors: after N new-error notifications in a
29
+ # window (per process) one summary message replaces the rest. Every error
30
+ # is still recorded.
31
+ # config.notification_cooldown_minutes = 5 # 0 = no cooldown
32
+ # config.notification_burst_limit = 10 # 0 = no cap
33
+ # config.notification_burst_window_seconds = 60
24
34
 
25
35
  # === Custom Authentication (optional) ===
26
36
  # Use your app's existing auth instead of HTTP Basic Auth.
@@ -61,7 +71,8 @@ RailsErrorDashboard.configure do |config|
61
71
  # User model for error associations
62
72
  config.user_model = "User"
63
73
 
64
- # Error retention policy (days to keep errors before automatic deletion)
74
+ # Error retention policy: an error is deleted once it has not been seen for
75
+ # this many days (an error that is still occurring is never deleted)
65
76
  # Set to nil to keep errors forever (not recommended for production)
66
77
  # Run cleanup manually: rails error_dashboard:retention_cleanup
67
78
  # Or schedule the job: RailsErrorDashboard::RetentionCleanupJob.perform_later
@@ -4,7 +4,11 @@ module RailsErrorDashboard
4
4
  module Commands
5
5
  # Command: Assign an error to a user
6
6
  # This is a write operation that updates assignment fields on an ErrorLog record
7
+ # Returns {success: bool, error: ErrorLog}; a failure also carries
8
+ # reason: :blank_assignee and writes nothing.
7
9
  class AssignError
10
+ MAX_ASSIGNEE_LENGTH = 255
11
+
8
12
  def self.call(error_id, assigned_to:)
9
13
  new(error_id, assigned_to).call
10
14
  end
@@ -16,12 +20,18 @@ module RailsErrorDashboard
16
20
 
17
21
  def call
18
22
  error = ErrorLog.find(@error_id)
23
+
24
+ # A blank name used to "assign" the error to nobody and still move it to
25
+ # in_progress. A nested parameter is not a name either.
26
+ assignee = @assigned_to.is_a?(String) ? @assigned_to.strip.presence : nil
27
+ return { success: false, error: error, reason: :blank_assignee } unless assignee
28
+
19
29
  error.update!(
20
- assigned_to: @assigned_to,
30
+ assigned_to: assignee.truncate(MAX_ASSIGNEE_LENGTH, omission: ""),
21
31
  assigned_at: Time.current,
22
32
  status: "in_progress" # Auto-transition to in_progress when assigned
23
33
  )
24
- error
34
+ { success: true, error: error }
25
35
  end
26
36
  end
27
37
  end
@@ -39,6 +39,8 @@ module RailsErrorDashboard
39
39
  updated += ErrorLog.where(id: id, environment: nil).update_all(environment: env)
40
40
  end
41
41
  end
42
+ # Analytics groups by environment, and its result is cached.
43
+ Services::AnalyticsCacheManager.clear if updated.positive?
42
44
  updated
43
45
  end
44
46
 
@@ -0,0 +1,42 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RailsErrorDashboard
4
+ module Commands
5
+ # Command: give resolved errors that have no resolved_at a best-known one
6
+ #
7
+ # Errors resolved through the status workflow before 0.13.0 were marked
8
+ # resolved without a resolved_at, so MTTR never counted them. The moment of
9
+ # resolution was not recorded anywhere, but it was the row's last write
10
+ # unless something touched it afterwards, so updated_at is the closest
11
+ # available value.
12
+ #
13
+ # Behind `rake error_dashboard:backfill_resolved_at`. Idempotent: it only
14
+ # selects rows whose resolved_at is still NULL.
15
+ #
16
+ # @example
17
+ # BackfillResolvedAt.call # => { updated: 42 }
18
+ class BackfillResolvedAt
19
+ BATCH_SIZE = 1000
20
+
21
+ def self.call(batch_size: BATCH_SIZE)
22
+ new(batch_size: batch_size).call
23
+ end
24
+
25
+ def initialize(batch_size: BATCH_SIZE)
26
+ @batch_size = batch_size
27
+ end
28
+
29
+ def call
30
+ updated = 0
31
+
32
+ ErrorLog.where(resolved: true, resolved_at: nil).in_batches(of: @batch_size) do |batch|
33
+ # SQL column reference, not a Ruby value: each row gets its own
34
+ # updated_at. update_all leaves updated_at itself alone.
35
+ updated += batch.update_all("resolved_at = updated_at")
36
+ end
37
+
38
+ { updated: updated }
39
+ end
40
+ end
41
+ end
42
+ end
@@ -23,6 +23,7 @@ module RailsErrorDashboard
23
23
  error_ids_to_delete = errors.pluck(:id)
24
24
 
25
25
  errors.destroy_all
26
+ Services::AnalyticsCacheManager.clear
26
27
 
27
28
  # Dispatch plugin event for batch deleted errors
28
29
  PluginRegistry.dispatch(:on_errors_batch_deleted, error_ids_to_delete) if error_ids_to_delete.any?
@@ -41,6 +41,8 @@ module RailsErrorDashboard
41
41
  end
42
42
  end
43
43
 
44
+ Services::AnalyticsCacheManager.clear if muted_count.positive?
45
+
44
46
  PluginRegistry.dispatch(:on_errors_batch_muted, muted_errors) if muted_errors.any?
45
47
 
46
48
  {
@@ -45,6 +45,8 @@ module RailsErrorDashboard
45
45
  end
46
46
 
47
47
  # Dispatch plugin event for batch resolved errors
48
+ Services::AnalyticsCacheManager.clear if resolved_count.positive?
49
+
48
50
  PluginRegistry.dispatch(:on_errors_batch_resolved, resolved_errors) if resolved_errors.any?
49
51
 
50
52
  {
@@ -39,6 +39,8 @@ module RailsErrorDashboard
39
39
  end
40
40
  end
41
41
 
42
+ Services::AnalyticsCacheManager.clear if unmuted_count.positive?
43
+
42
44
  PluginRegistry.dispatch(:on_errors_batch_unmuted, unmuted_errors) if unmuted_errors.any?
43
45
 
44
46
  {