rails_error_dashboard 0.12.1 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/app/controllers/rails_error_dashboard/application_controller.rb +92 -11
- data/app/controllers/rails_error_dashboard/errors_controller.rb +111 -34
- data/app/controllers/rails_error_dashboard/webhooks_controller.rb +3 -0
- data/app/helpers/rails_error_dashboard/application_helper.rb +46 -1
- data/app/helpers/rails_error_dashboard/backtrace_helper.rb +8 -0
- data/app/jobs/rails_error_dashboard/async_error_logging_job.rb +10 -0
- data/app/jobs/rails_error_dashboard/concerns/plain_channel_message.rb +71 -0
- data/app/jobs/rails_error_dashboard/notification_burst_summary_job.rb +55 -0
- data/app/jobs/rails_error_dashboard/retention_cleanup_job.rb +122 -1
- data/app/jobs/rails_error_dashboard/storm_flush_job.rb +7 -4
- data/app/jobs/rails_error_dashboard/storm_notification_job.rb +8 -44
- data/app/models/rails_error_dashboard/error_baseline.rb +12 -9
- data/app/models/rails_error_dashboard/error_comment.rb +0 -5
- data/app/models/rails_error_dashboard/error_log.rb +19 -3
- data/app/models/rails_error_dashboard/error_logs_record.rb +34 -0
- data/app/models/rails_error_dashboard/error_occurrence.rb +9 -1
- data/app/models/rails_error_dashboard/event_count.rb +132 -0
- data/app/models/rails_error_dashboard/event_timing_gap.rb +55 -0
- data/app/views/layouts/rails_error_dashboard.html.erb +58 -5
- data/app/views/rails_error_dashboard/errors/_discussion.html.erb +18 -11
- data/app/views/rails_error_dashboard/errors/_issue_section.html.erb +22 -8
- data/app/views/rails_error_dashboard/errors/_request_context.html.erb +2 -0
- data/app/views/rails_error_dashboard/errors/_sidebar_metadata.html.erb +15 -6
- data/app/views/rails_error_dashboard/errors/analytics.html.erb +8 -8
- data/app/views/rails_error_dashboard/errors/diagnostic_dumps.html.erb +1 -1
- data/app/views/rails_error_dashboard/errors/index.html.erb +6 -1
- data/app/views/rails_error_dashboard/errors/overview.html.erb +12 -0
- data/app/views/rails_error_dashboard/errors/platform_comparison.html.erb +7 -7
- data/app/views/rails_error_dashboard/errors/releases.html.erb +2 -2
- data/app/views/rails_error_dashboard/errors/settings.html.erb +2 -0
- data/app/views/rails_error_dashboard/errors/show.html.erb +3 -3
- data/config/locales/de.yml +30 -0
- data/config/locales/en.yml +37 -0
- data/config/locales/es.yml +30 -0
- data/config/locales/fr.yml +30 -0
- data/config/locales/it.yml +30 -0
- data/config/locales/ja.yml +30 -0
- data/config/locales/pl.yml +30 -0
- data/config/locales/pt-BR.yml +30 -0
- data/config/locales/ru.yml +30 -0
- data/config/locales/uk.yml +30 -0
- data/config/locales/zh-CN.yml +30 -0
- data/db/migrate/20260917000001_add_last_notified_at_to_error_logs.rb +40 -0
- data/db/migrate/20260919000001_create_event_counts.rb +71 -0
- data/db/migrate/20260920000001_add_buckets_incomplete_to_storm_events.rb +25 -0
- data/db/migrate/20260920000002_create_event_timing_gaps.rb +55 -0
- data/lib/generators/rails_error_dashboard/install/templates/initializer.rb +12 -1
- data/lib/rails_error_dashboard/commands/assign_error.rb +12 -2
- data/lib/rails_error_dashboard/commands/backfill_environments.rb +2 -0
- data/lib/rails_error_dashboard/commands/backfill_resolved_at.rb +42 -0
- data/lib/rails_error_dashboard/commands/batch_delete_errors.rb +1 -0
- data/lib/rails_error_dashboard/commands/batch_mute_errors.rb +2 -0
- data/lib/rails_error_dashboard/commands/batch_resolve_errors.rb +2 -0
- data/lib/rails_error_dashboard/commands/batch_unmute_errors.rb +2 -0
- data/lib/rails_error_dashboard/commands/find_or_increment_error.rb +109 -11
- data/lib/rails_error_dashboard/commands/flush_rack_attack_events.rb +3 -1
- data/lib/rails_error_dashboard/commands/flush_storm_counts.rb +252 -12
- data/lib/rails_error_dashboard/commands/flush_swallowed_exceptions.rb +5 -2
- data/lib/rails_error_dashboard/commands/link_existing_issue.rb +1 -0
- data/lib/rails_error_dashboard/commands/log_error.rb +291 -40
- data/lib/rails_error_dashboard/commands/mute_error.rb +1 -0
- data/lib/rails_error_dashboard/commands/resolve_error.rb +2 -0
- data/lib/rails_error_dashboard/commands/scrub_invalid_encoding.rb +104 -0
- data/lib/rails_error_dashboard/commands/snooze_error.rb +34 -10
- data/lib/rails_error_dashboard/commands/unmute_error.rb +1 -0
- data/lib/rails_error_dashboard/commands/update_error_priority.rb +23 -2
- data/lib/rails_error_dashboard/commands/update_error_status.rb +33 -6
- data/lib/rails_error_dashboard/configuration.rb +39 -1
- data/lib/rails_error_dashboard/engine.rb +28 -0
- data/lib/rails_error_dashboard/manual_error_reporter.rb +16 -5
- data/lib/rails_error_dashboard/queries/analytics_stats.rb +89 -30
- data/lib/rails_error_dashboard/queries/baseline_stats.rb +107 -0
- data/lib/rails_error_dashboard/queries/dashboard_stats.rb +216 -74
- data/lib/rails_error_dashboard/queries/error_correlation.rb +6 -3
- data/lib/rails_error_dashboard/queries/errors_list.rb +15 -2
- data/lib/rails_error_dashboard/queries/event_volume.rb +503 -0
- data/lib/rails_error_dashboard/queries/similar_errors.rb +1 -1
- data/lib/rails_error_dashboard/services/analytics_cache_manager.rb +43 -18
- data/lib/rails_error_dashboard/services/backtrace_processor.rb +3 -1
- data/lib/rails_error_dashboard/services/breadcrumb_collector.rb +23 -0
- data/lib/rails_error_dashboard/services/cause_chain_extractor.rb +3 -1
- data/lib/rails_error_dashboard/services/codeberg_issue_client.rb +13 -2
- data/lib/rails_error_dashboard/services/diagnostic_dump_generator.rb +5 -3
- data/lib/rails_error_dashboard/services/encoding_sanitizer.rb +80 -0
- data/lib/rails_error_dashboard/services/error_broadcaster.rb +180 -32
- data/lib/rails_error_dashboard/services/error_hash_generator.rb +9 -3
- data/lib/rails_error_dashboard/services/error_notification_dispatcher.rb +13 -0
- data/lib/rails_error_dashboard/services/exception_filter.rb +55 -0
- data/lib/rails_error_dashboard/services/git_head_reader.rb +102 -0
- data/lib/rails_error_dashboard/services/notification_throttler.rb +172 -28
- data/lib/rails_error_dashboard/services/sensitive_data_filter.rb +47 -1
- data/lib/rails_error_dashboard/services/storm_protection/circuit_breaker.rb +59 -3
- data/lib/rails_error_dashboard/services/storm_protection/count_buffer.rb +64 -6
- data/lib/rails_error_dashboard/services/storm_protection/fingerprint_buckets.rb +25 -2
- data/lib/rails_error_dashboard/services/storm_protection/gate.rb +32 -5
- data/lib/rails_error_dashboard/services/swallowed_exception_tracker.rb +99 -23
- data/lib/rails_error_dashboard/services/url_safety.rb +40 -0
- data/lib/rails_error_dashboard/services/variable_serializer.rb +125 -10
- data/lib/rails_error_dashboard/subscribers/breadcrumb_subscriber.rb +111 -0
- data/lib/rails_error_dashboard/subscribers/issue_tracker_subscriber.rb +11 -2
- data/lib/rails_error_dashboard/value_objects/error_context.rb +40 -3
- data/lib/rails_error_dashboard/version.rb +1 -1
- data/lib/rails_error_dashboard.rb +34 -0
- data/lib/tasks/error_dashboard.rake +54 -4
- metadata +16 -2
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RailsErrorDashboard
|
|
4
|
+
# Sends the ONE message that replaces the rest of a burst of new-error
|
|
5
|
+
# notifications.
|
|
6
|
+
#
|
|
7
|
+
# A bad deploy can produce hundreds of DISTINCT new errors. Each is a first
|
|
8
|
+
# occurrence, so the per-error cooldown never applies; without a cap every
|
|
9
|
+
# one of them pages. NotificationThrottler.burst_decision lets the first
|
|
10
|
+
# config.notification_burst_limit through per window and asks for this job
|
|
11
|
+
# exactly once, when the limit is first exceeded.
|
|
12
|
+
#
|
|
13
|
+
# The message states the limit and the window rather than a count of what
|
|
14
|
+
# was suppressed: it is sent when suppression STARTS, and a count at that
|
|
15
|
+
# moment would always be one. Every error is still stored and counted; only
|
|
16
|
+
# the notifications are held back.
|
|
17
|
+
class NotificationBurstSummaryJob < ApplicationJob
|
|
18
|
+
include Concerns::PlainChannelMessage
|
|
19
|
+
|
|
20
|
+
queue_as :default
|
|
21
|
+
|
|
22
|
+
# @param limit [Integer] config.notification_burst_limit when the cap engaged
|
|
23
|
+
# @param window_seconds [Integer] config.notification_burst_window_seconds
|
|
24
|
+
# @param locale [String, nil] resolved at enqueue time
|
|
25
|
+
def perform(limit:, window_seconds:, locale: nil)
|
|
26
|
+
config = RailsErrorDashboard.configuration
|
|
27
|
+
|
|
28
|
+
deliver_plain_message(build_message(limit, window_seconds, config, job_locale(locale)), {
|
|
29
|
+
event: "new_error_notifications_suppressed",
|
|
30
|
+
limit: limit,
|
|
31
|
+
window_seconds: window_seconds,
|
|
32
|
+
application: app_name(config)
|
|
33
|
+
})
|
|
34
|
+
rescue => e
|
|
35
|
+
Rails.logger.error("[RailsErrorDashboard] NotificationBurstSummaryJob failed: #{e.class} - #{e.message}")
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
private
|
|
39
|
+
|
|
40
|
+
# Whole sentences joined, never fragments: see StormNotificationJob.
|
|
41
|
+
# ":warning:" is Slack/Discord emoji shortcode, not text.
|
|
42
|
+
def build_message(limit, window_seconds, config, locale)
|
|
43
|
+
dashboard = (config.dashboard_base_url || "").chomp("/")
|
|
44
|
+
link = if dashboard.present?
|
|
45
|
+
" " + t("red.notifications.burst.dashboard", locale, url: "#{dashboard}/errors")
|
|
46
|
+
else
|
|
47
|
+
""
|
|
48
|
+
end
|
|
49
|
+
|
|
50
|
+
":warning: " \
|
|
51
|
+
"#{t("red.notifications.burst.suppressed", locale, application: app_name(config), limit: limit, window: window_seconds)} " \
|
|
52
|
+
"#{t("red.notifications.burst.still_recorded", locale)}#{link}"
|
|
53
|
+
end
|
|
54
|
+
end
|
|
55
|
+
end
|
|
@@ -14,6 +14,25 @@ module RailsErrorDashboard
|
|
|
14
14
|
class RetentionCleanupJob < ApplicationJob
|
|
15
15
|
queue_as :default
|
|
16
16
|
|
|
17
|
+
# The errors retention applies to: "not seen for retention_days", not
|
|
18
|
+
# "first seen retention_days ago". occurred_at is stamped when a group is
|
|
19
|
+
# created and never moves, so expiring by it deleted errors that were still
|
|
20
|
+
# happening today -- with their comments and triage history -- the day
|
|
21
|
+
# they turned N days old.
|
|
22
|
+
#
|
|
23
|
+
# Equivalent to COALESCE(last_seen_at, occurred_at) < cutoff (the NULL arm
|
|
24
|
+
# covers rows from before last_seen_at existed), but written so that each
|
|
25
|
+
# arm can use its own index; wrapping the columns in COALESCE would force a
|
|
26
|
+
# full scan of the largest table on every run.
|
|
27
|
+
#
|
|
28
|
+
# Public because the rake task previews the same selection before it asks
|
|
29
|
+
# for confirmation.
|
|
30
|
+
def self.expired_scope(cutoff)
|
|
31
|
+
ErrorLog.where(
|
|
32
|
+
"last_seen_at < :cutoff OR (last_seen_at IS NULL AND occurred_at < :cutoff)", cutoff: cutoff
|
|
33
|
+
)
|
|
34
|
+
end
|
|
35
|
+
|
|
17
36
|
# @return [Integer] number of errors deleted
|
|
18
37
|
def perform
|
|
19
38
|
retention_days = RailsErrorDashboard.configuration.retention_days
|
|
@@ -26,8 +45,11 @@ module RailsErrorDashboard
|
|
|
26
45
|
# whenever no error logs happen to be expired.
|
|
27
46
|
cleanup_rack_attack_events(cutoff)
|
|
28
47
|
cleanup_storm_flush_batches(cutoff)
|
|
48
|
+
cleanup_diagnostic_dumps(cutoff)
|
|
49
|
+
cleanup_swallowed_exceptions(cutoff)
|
|
50
|
+
cleanup_event_timing_gaps(cutoff)
|
|
29
51
|
|
|
30
|
-
expired_scope =
|
|
52
|
+
expired_scope = self.class.expired_scope(cutoff)
|
|
31
53
|
return 0 if expired_scope.none?
|
|
32
54
|
|
|
33
55
|
deleted_count = 0
|
|
@@ -37,6 +59,12 @@ module RailsErrorDashboard
|
|
|
37
59
|
|
|
38
60
|
# Batch delete dependent records (occurrences, comments, cascade patterns)
|
|
39
61
|
ErrorOccurrence.where(error_log_id: expired_ids_scope).in_batches(of: 1000).delete_all
|
|
62
|
+
# Hour buckets for storm-shed events. Listed explicitly because this job
|
|
63
|
+
# deletes with delete_all, which does not fire the has_many :dependent
|
|
64
|
+
# callback on ErrorLog -- without this line the buckets outlive the group
|
|
65
|
+
# they describe, forever. The migration's own comment promised this
|
|
66
|
+
# pruning before the code existed.
|
|
67
|
+
EventCount.where(error_log_id: expired_ids_scope).in_batches(of: 1000).delete_all if EventCount.table_exists?
|
|
40
68
|
ErrorComment.where(error_log_id: expired_ids_scope).in_batches(of: 1000).delete_all
|
|
41
69
|
CascadePattern.where(parent_error_id: expired_ids_scope)
|
|
42
70
|
.or(CascadePattern.where(child_error_id: expired_ids_scope))
|
|
@@ -48,6 +76,9 @@ module RailsErrorDashboard
|
|
|
48
76
|
deleted_count += batch_size
|
|
49
77
|
end
|
|
50
78
|
|
|
79
|
+
# delete_all skips callbacks, and the stat cards are cached.
|
|
80
|
+
Services::AnalyticsCacheManager.clear if deleted_count > 0
|
|
81
|
+
|
|
51
82
|
if deleted_count > 0
|
|
52
83
|
RailsErrorDashboard::Logger.info(
|
|
53
84
|
"[RailsErrorDashboard] Retention cleanup: deleted #{deleted_count} errors older than #{retention_days} days"
|
|
@@ -86,6 +117,96 @@ module RailsErrorDashboard
|
|
|
86
117
|
)
|
|
87
118
|
end
|
|
88
119
|
|
|
120
|
+
# Diagnostic dumps and swallowed-exception aggregates are written
|
|
121
|
+
# independently of error logs and nothing else ever deletes them, so they
|
|
122
|
+
# grew without bound. Not gated on their feature flags: rows written while
|
|
123
|
+
# a feature was on must still expire after it is switched off. Own rescue
|
|
124
|
+
# each, like the cleanups around them.
|
|
125
|
+
def cleanup_diagnostic_dumps(cutoff)
|
|
126
|
+
return unless DiagnosticDump.table_exists?
|
|
127
|
+
|
|
128
|
+
deleted = 0
|
|
129
|
+
DiagnosticDump.where("captured_at < ?", cutoff).in_batches(of: 1000) do |batch|
|
|
130
|
+
deleted += batch.delete_all
|
|
131
|
+
end
|
|
132
|
+
|
|
133
|
+
if deleted > 0
|
|
134
|
+
RailsErrorDashboard::Logger.info(
|
|
135
|
+
"[RailsErrorDashboard] Retention cleanup: deleted #{deleted} diagnostic dumps"
|
|
136
|
+
)
|
|
137
|
+
end
|
|
138
|
+
rescue => e
|
|
139
|
+
RailsErrorDashboard::Logger.debug(
|
|
140
|
+
"[RailsErrorDashboard] Diagnostic dump retention cleanup failed: #{e.class} - #{e.message}"
|
|
141
|
+
)
|
|
142
|
+
end
|
|
143
|
+
|
|
144
|
+
def cleanup_swallowed_exceptions(cutoff)
|
|
145
|
+
return unless SwallowedException.table_exists?
|
|
146
|
+
|
|
147
|
+
deleted = 0
|
|
148
|
+
SwallowedException.where("period_hour < ?", cutoff).in_batches(of: 1000) do |batch|
|
|
149
|
+
deleted += batch.delete_all
|
|
150
|
+
end
|
|
151
|
+
|
|
152
|
+
if deleted > 0
|
|
153
|
+
RailsErrorDashboard::Logger.info(
|
|
154
|
+
"[RailsErrorDashboard] Retention cleanup: deleted #{deleted} swallowed exception records"
|
|
155
|
+
)
|
|
156
|
+
end
|
|
157
|
+
rescue => e
|
|
158
|
+
RailsErrorDashboard::Logger.debug(
|
|
159
|
+
"[RailsErrorDashboard] Swallowed exception retention cleanup failed: #{e.class} - #{e.message}"
|
|
160
|
+
)
|
|
161
|
+
end
|
|
162
|
+
|
|
163
|
+
# Timing gaps are pruned by their OWN age, not by a group.
|
|
164
|
+
#
|
|
165
|
+
# A gap describes an interval, not an error, so there is no error_log_id to
|
|
166
|
+
# cascade from -- without this it would accumulate for the life of the
|
|
167
|
+
# installation. It also runs ABOVE the early return in #perform, which
|
|
168
|
+
# fires whenever no error logs happen to be expired: gaps expire
|
|
169
|
+
# independently of errors, exactly like the rack-attack events whose
|
|
170
|
+
# comment already warns about this.
|
|
171
|
+
#
|
|
172
|
+
# Safe to prune on covered_until: once the cutoff has moved past a gap, no
|
|
173
|
+
# window the dashboard displays can still reach it, so the warning it
|
|
174
|
+
# carries is no longer meaningful. Own rescue, like its siblings.
|
|
175
|
+
def cleanup_event_timing_gaps(cutoff)
|
|
176
|
+
return unless EventTimingGap.table_exists?
|
|
177
|
+
|
|
178
|
+
# Kept for the LONGER of the two horizons that govern it.
|
|
179
|
+
#
|
|
180
|
+
# This used the configured retention cutoff alone, which is wrong
|
|
181
|
+
# whenever retention is shorter than the window the dashboard reports on:
|
|
182
|
+
# at retention_days = 7 the gap was deleted while the events it qualified
|
|
183
|
+
# were still on the page -- ten monthly events, no warning, group still
|
|
184
|
+
# active. A gap outlives its own retention precisely because the figures
|
|
185
|
+
# it qualifies do.
|
|
186
|
+
#
|
|
187
|
+
# Deleting it only once BOTH clocks have passed means the warning can
|
|
188
|
+
# never disappear while the numbers it describes are still displayed.
|
|
189
|
+
# The reverse case is unaffected: with the 90-day default the retention
|
|
190
|
+
# cutoff is already the later of the two, so nothing is kept longer than
|
|
191
|
+
# before.
|
|
192
|
+
gap_cutoff = [ cutoff, Queries::DashboardStats::WIDEST_DISPLAYED_WINDOW.ago ].min
|
|
193
|
+
|
|
194
|
+
deleted = 0
|
|
195
|
+
EventTimingGap.where("covered_until < ?", gap_cutoff).in_batches(of: 1000) do |batch|
|
|
196
|
+
deleted += batch.delete_all
|
|
197
|
+
end
|
|
198
|
+
|
|
199
|
+
if deleted > 0
|
|
200
|
+
RailsErrorDashboard::Logger.info(
|
|
201
|
+
"[RailsErrorDashboard] Retention cleanup: deleted #{deleted} event timing gaps"
|
|
202
|
+
)
|
|
203
|
+
end
|
|
204
|
+
rescue => e
|
|
205
|
+
RailsErrorDashboard::Logger.debug(
|
|
206
|
+
"[RailsErrorDashboard] Event timing gap retention cleanup failed: #{e.class} - #{e.message}"
|
|
207
|
+
)
|
|
208
|
+
end
|
|
209
|
+
|
|
89
210
|
# Expire aggregated Rack Attack event rows. Isolated in its own rescue so a
|
|
90
211
|
# failure here (e.g. table not yet migrated) never blocks error cleanup.
|
|
91
212
|
def cleanup_rack_attack_events(cutoff)
|
|
@@ -21,11 +21,14 @@ module RailsErrorDashboard
|
|
|
21
21
|
entries: entries, overflow: overflow, episode: episode, batch_id: batch_id
|
|
22
22
|
)
|
|
23
23
|
|
|
24
|
-
# A batch that
|
|
25
|
-
#
|
|
26
|
-
#
|
|
24
|
+
# A batch that wrote nothing is not a delivered batch. Fail the job so
|
|
25
|
+
# Active Job retries it rather than dropping counts that were only ever
|
|
26
|
+
# held in one process's memory. Two shapes reach here: every entry failed
|
|
27
|
+
# permanently, and a transient store failure that rolled the whole batch
|
|
28
|
+
# back (retryable: true) -- the latter is intact and safe to replay.
|
|
27
29
|
if result.is_a?(Hash) && result[:success] == false
|
|
28
|
-
|
|
30
|
+
reason = result[:retryable] ? "storm flush rolled back" : "storm flush reconciled nothing"
|
|
31
|
+
raise FlushFailed, "#{reason}: #{result[:error]}"
|
|
29
32
|
end
|
|
30
33
|
|
|
31
34
|
result
|
|
@@ -7,6 +7,8 @@ module RailsErrorDashboard
|
|
|
7
7
|
# help nobody) — this one message replaces them. The gate guarantees at
|
|
8
8
|
# most one enqueue per episode; this job just delivers.
|
|
9
9
|
class StormNotificationJob < ApplicationJob
|
|
10
|
+
include Concerns::PlainChannelMessage
|
|
11
|
+
|
|
10
12
|
queue_as :default
|
|
11
13
|
|
|
12
14
|
# @param started_at [String] ISO8601 episode start
|
|
@@ -17,23 +19,12 @@ module RailsErrorDashboard
|
|
|
17
19
|
config = RailsErrorDashboard.configuration
|
|
18
20
|
message = build_message(started_at, state, config, job_locale(locale))
|
|
19
21
|
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
end
|
|
27
|
-
|
|
28
|
-
if config.enable_webhook_notifications && config.webhook_urls.any?
|
|
29
|
-
payload = {
|
|
30
|
-
event: "error_storm_detected",
|
|
31
|
-
started_at: started_at,
|
|
32
|
-
state: state,
|
|
33
|
-
application: app_name(config)
|
|
34
|
-
}
|
|
35
|
-
config.webhook_urls.each { |url| post_json(url, payload) }
|
|
36
|
-
end
|
|
22
|
+
deliver_plain_message(message, {
|
|
23
|
+
event: "error_storm_detected",
|
|
24
|
+
started_at: started_at,
|
|
25
|
+
state: state,
|
|
26
|
+
application: app_name(config)
|
|
27
|
+
})
|
|
37
28
|
rescue => e
|
|
38
29
|
Rails.logger.error("[RailsErrorDashboard] StormNotificationJob failed: #{e.class} - #{e.message}")
|
|
39
30
|
end
|
|
@@ -61,32 +52,5 @@ module RailsErrorDashboard
|
|
|
61
52
|
"#{t("red.notifications.storm.engaged", locale, mode: mode)} " \
|
|
62
53
|
"#{t("red.notifications.storm.suppressed", locale)}#{link}"
|
|
63
54
|
end
|
|
64
|
-
|
|
65
|
-
def t(key, locale, **options)
|
|
66
|
-
RailsErrorDashboard::I18nStore.translate(key, locale: locale, **options)
|
|
67
|
-
end
|
|
68
|
-
|
|
69
|
-
def app_name(config)
|
|
70
|
-
config.application_name || ENV["APPLICATION_NAME"] ||
|
|
71
|
-
(defined?(Rails) && Rails.application.class.module_parent_name) || "Rails Application"
|
|
72
|
-
end
|
|
73
|
-
|
|
74
|
-
def post_json(url, payload)
|
|
75
|
-
if defined?(HTTParty)
|
|
76
|
-
HTTParty.post(url, body: payload.to_json,
|
|
77
|
-
headers: { "Content-Type" => "application/json" }, timeout: 10)
|
|
78
|
-
else
|
|
79
|
-
uri = URI(url)
|
|
80
|
-
http = Net::HTTP.new(uri.host, uri.port)
|
|
81
|
-
http.use_ssl = uri.scheme == "https"
|
|
82
|
-
http.open_timeout = 5
|
|
83
|
-
http.read_timeout = 10
|
|
84
|
-
request = Net::HTTP::Post.new(uri.path, { "Content-Type" => "application/json" })
|
|
85
|
-
request.body = payload.to_json
|
|
86
|
-
http.request(request)
|
|
87
|
-
end
|
|
88
|
-
rescue => e
|
|
89
|
-
Rails.logger.error("[RailsErrorDashboard] Storm notification post failed: #{e.message}")
|
|
90
|
-
end
|
|
91
55
|
end
|
|
92
56
|
end
|
|
@@ -48,17 +48,20 @@ module RailsErrorDashboard
|
|
|
48
48
|
return nil if mean.nil? || std_dev.nil?
|
|
49
49
|
return nil if current_count <= mean
|
|
50
50
|
|
|
51
|
-
|
|
51
|
+
# nil when std_dev is zero: a perfectly flat history has no spread to
|
|
52
|
+
# measure against, and dividing by it made every count :critical.
|
|
53
|
+
sigma = std_devs_above_mean(current_count)
|
|
54
|
+
return nil if sigma.nil?
|
|
52
55
|
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
:high
|
|
58
|
-
when (sensitivity + 2)..Float::INFINITY
|
|
56
|
+
# Half-open bands, highest first: a boundary value belongs to the band it
|
|
57
|
+
# opens (3.0 is :high, 4.0 is :critical). Inclusive ranges matched
|
|
58
|
+
# top-down gave each boundary to the band below it.
|
|
59
|
+
if sigma >= sensitivity + 2
|
|
59
60
|
:critical
|
|
60
|
-
|
|
61
|
-
|
|
61
|
+
elsif sigma >= sensitivity + 1
|
|
62
|
+
:high
|
|
63
|
+
elsif sigma >= sensitivity
|
|
64
|
+
:elevated
|
|
62
65
|
end
|
|
63
66
|
end
|
|
64
67
|
|
|
@@ -14,11 +14,6 @@ module RailsErrorDashboard
|
|
|
14
14
|
scope :recent_first, -> { order(created_at: :desc) }
|
|
15
15
|
scope :oldest_first, -> { order(created_at: :asc) }
|
|
16
16
|
|
|
17
|
-
# Get formatted timestamp for display
|
|
18
|
-
def formatted_time
|
|
19
|
-
created_at.strftime("%b %d, %Y at %I:%M %p")
|
|
20
|
-
end
|
|
21
|
-
|
|
22
17
|
# Check if comment was created recently (within last hour)
|
|
23
18
|
def recent?
|
|
24
19
|
created_at > 1.hour.ago
|
|
@@ -37,6 +37,16 @@ module RailsErrorDashboard
|
|
|
37
37
|
# Association for tracking individual error occurrences
|
|
38
38
|
has_many :error_occurrences, class_name: "RailsErrorDashboard::ErrorOccurrence", dependent: :destroy
|
|
39
39
|
|
|
40
|
+
# Hour buckets for storm-shed events. delete_all, not destroy: these rows
|
|
41
|
+
# are pure counters with no callbacks, and a group can own one per hour.
|
|
42
|
+
#
|
|
43
|
+
# The association has to live HERE. `dependent:` on EventCount's own
|
|
44
|
+
# belongs_to is rejected by Rails (":dependent option must be one of
|
|
45
|
+
# [:destroy, :delete, :destroy_async]"), so the cleanup can only be
|
|
46
|
+
# declared from the parent side. Retention deletes them separately as well,
|
|
47
|
+
# because it uses delete_all, which does not fire callbacks.
|
|
48
|
+
has_many :event_counts, class_name: "RailsErrorDashboard::EventCount", dependent: :delete_all
|
|
49
|
+
|
|
40
50
|
# Comments used as internal audit trail for workflow actions (snooze, mute, status changes).
|
|
41
51
|
# Manual comment form removed in v0.6 — discussion now lives on issue tracker.
|
|
42
52
|
has_many :comments, class_name: "RailsErrorDashboard::ErrorComment", foreign_key: :error_log_id, dependent: :destroy
|
|
@@ -86,9 +96,9 @@ module RailsErrorDashboard
|
|
|
86
96
|
after_create_commit -> { Services::ErrorBroadcaster.broadcast_new(self) }
|
|
87
97
|
after_update_commit -> { Services::ErrorBroadcaster.broadcast_update(self) }
|
|
88
98
|
|
|
89
|
-
#
|
|
90
|
-
|
|
91
|
-
|
|
99
|
+
# No cache invalidation here, on purpose. A save is what a CAPTURE does, in
|
|
100
|
+
# the host app's request thread; the stats caches expire by TTL instead, and
|
|
101
|
+
# the commands behind user actions call AnalyticsCacheManager.clear themselves.
|
|
92
102
|
|
|
93
103
|
def set_defaults
|
|
94
104
|
self.platform ||= "API"
|
|
@@ -271,7 +281,13 @@ module RailsErrorDashboard
|
|
|
271
281
|
end
|
|
272
282
|
end
|
|
273
283
|
|
|
284
|
+
# The five workflow statuses. One list, so that a command can tell
|
|
285
|
+
# "unknown status" from "known, but not reachable from here".
|
|
286
|
+
STATUSES = %w[new in_progress investigating resolved wont_fix].freeze
|
|
287
|
+
|
|
274
288
|
def can_transition_to?(new_status)
|
|
289
|
+
return false unless STATUSES.include?(new_status)
|
|
290
|
+
|
|
275
291
|
# Define valid status transitions
|
|
276
292
|
valid_transitions = {
|
|
277
293
|
"new" => [ "in_progress", "investigating", "wont_fix" ],
|
|
@@ -35,6 +35,23 @@ module RailsErrorDashboard
|
|
|
35
35
|
# after the user's configuration is loaded
|
|
36
36
|
# See lib/rails_error_dashboard/engine.rb
|
|
37
37
|
|
|
38
|
+
# Read-side half of the invalid-bytes defence. Capture scrubs what it
|
|
39
|
+
# writes, but SQLite and MySQL will happily hold a row written before that
|
|
40
|
+
# (or by anything else), and one such string takes down every page that
|
|
41
|
+
# renders it: ERB raises "invalid byte sequence in UTF-8". Scrubbing on load
|
|
42
|
+
# makes the page render whether or not the operator has run
|
|
43
|
+
# error_dashboard:scrub_invalid_encoding yet. In memory only: nothing is
|
|
44
|
+
# written, and the record is not left dirty.
|
|
45
|
+
after_find :scrub_invalid_strings
|
|
46
|
+
|
|
47
|
+
# Names of the attributes the load-time scrub had to repair, so that
|
|
48
|
+
# Commands::ScrubInvalidEncoding can persist the repair: once the values
|
|
49
|
+
# are clean in memory, re-reading them can no longer tell a bad row apart.
|
|
50
|
+
# @return [Array<String>]
|
|
51
|
+
def invalid_encoding_attributes
|
|
52
|
+
@invalid_encoding_attributes || []
|
|
53
|
+
end
|
|
54
|
+
|
|
38
55
|
# Rails' default for a string column when the adapter has one (MySQL).
|
|
39
56
|
DEFAULT_STRING_LIMIT = 255
|
|
40
57
|
|
|
@@ -68,5 +85,22 @@ module RailsErrorDashboard
|
|
|
68
85
|
limits[name] = column.limit || DEFAULT_STRING_LIMIT if column.type == :string
|
|
69
86
|
end
|
|
70
87
|
end
|
|
88
|
+
|
|
89
|
+
private
|
|
90
|
+
|
|
91
|
+
def scrub_invalid_strings
|
|
92
|
+
@attributes.each_value do |attribute|
|
|
93
|
+
value = attribute.value_before_type_cast
|
|
94
|
+
next unless value.is_a?(String)
|
|
95
|
+
next if value.encoding == Encoding::UTF_8 && value.valid_encoding? && !value.include?("\0")
|
|
96
|
+
|
|
97
|
+
name = attribute.name
|
|
98
|
+
self[name] = Services::EncodingSanitizer.scrub(value)
|
|
99
|
+
clear_attribute_changes([ name ])
|
|
100
|
+
(@invalid_encoding_attributes ||= []) << name
|
|
101
|
+
end
|
|
102
|
+
rescue => e
|
|
103
|
+
RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] scrub on load failed: #{e.class}: #{e.message}")
|
|
104
|
+
end
|
|
71
105
|
end
|
|
72
106
|
end
|
|
@@ -23,7 +23,15 @@ module RailsErrorDashboard
|
|
|
23
23
|
scope :in_time_window, ->(start_time, end_time) { where(occurred_at: start_time..end_time) }
|
|
24
24
|
scope :for_user, ->(user_id) { where(user_id: user_id) }
|
|
25
25
|
scope :for_request, ->(request_id) { where(request_id: request_id) }
|
|
26
|
-
|
|
26
|
+
# Takes the raw session ID (or a stored digest). Rows are stored as a keyed
|
|
27
|
+
# digest while filter_sensitive_data is on, and raw otherwise or before the
|
|
28
|
+
# upgrade, so both forms are matched. Blank matches nothing.
|
|
29
|
+
scope :for_session, lambda { |session_id|
|
|
30
|
+
raw = session_id.to_s
|
|
31
|
+
next none if raw.empty?
|
|
32
|
+
|
|
33
|
+
where(session_id: [ raw, Services::SensitiveDataFilter.digest_session_id(raw) ].compact.uniq)
|
|
34
|
+
}
|
|
27
35
|
|
|
28
36
|
# Find occurrences within a time window around this occurrence
|
|
29
37
|
# @param window_minutes [Integer] Time window in minutes (default: 5)
|
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RailsErrorDashboard
|
|
4
|
+
# How many storm-shed events landed on one error group in one hour.
|
|
5
|
+
#
|
|
6
|
+
# Storm protection sheds events by folding N of them into the group's
|
|
7
|
+
# occurrence_count and writing no ErrorOccurrence row. That keeps the total
|
|
8
|
+
# exact but leaves the events with no timestamp of their own, so a window
|
|
9
|
+
# query has nothing to filter them by. This table gives them one.
|
|
10
|
+
#
|
|
11
|
+
# Window volume is therefore:
|
|
12
|
+
#
|
|
13
|
+
# ErrorOccurrence rows in the window + EventCount buckets in the window
|
|
14
|
+
#
|
|
15
|
+
# Ordinary captures contribute the first term, shed events the second, and
|
|
16
|
+
# neither is double counted: an event that wrote an occurrence row is never
|
|
17
|
+
# also bucketed here.
|
|
18
|
+
#
|
|
19
|
+
# Inherits ErrorLogsRecord so separate-database routing applies.
|
|
20
|
+
class EventCount < ErrorLogsRecord
|
|
21
|
+
self.table_name = "rails_error_dashboard_event_counts"
|
|
22
|
+
|
|
23
|
+
belongs_to :error_log, class_name: "RailsErrorDashboard::ErrorLog", optional: true
|
|
24
|
+
|
|
25
|
+
scope :in_window, ->(from, to = nil) {
|
|
26
|
+
scope = where(arel_table[:bucket_at].gteq(from))
|
|
27
|
+
to ? scope.where(arel_table[:bucket_at].lt(to)) : scope
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
# Bucket width, shared with the PRODUCER.
|
|
31
|
+
#
|
|
32
|
+
# This must equal CountBuffer::BUCKET_SECONDS, and there is a spec that
|
|
33
|
+
# asserts it. The producer tallies 15-minute buckets precisely so that a
|
|
34
|
+
# local midnight falls on a bucket EDGE in every UTC offset in use --
|
|
35
|
+
# including +05:30 (Kolkata) and +05:45 (Kathmandu). Rounding those buckets
|
|
36
|
+
# to the hour here destroyed exactly the property the producer paid for:
|
|
37
|
+
# a storm event at 23:59:30 and one at 00:00:30 in Kolkata shared one
|
|
38
|
+
# bucket, and a day's total was wrong by the whole straddle.
|
|
39
|
+
BUCKET_SECONDS = Services::StormProtection::CountBuffer::BUCKET_SECONDS
|
|
40
|
+
|
|
41
|
+
# The bucket a time belongs to, in UTC. One definition, used by the writer
|
|
42
|
+
# and the readers -- a mismatch here silently splits or merges a bucket.
|
|
43
|
+
# @param time [Time]
|
|
44
|
+
# @return [Time]
|
|
45
|
+
def self.bucket_for(time)
|
|
46
|
+
time = (time || Time.current)
|
|
47
|
+
Time.at((time.to_i / BUCKET_SECONDS) * BUCKET_SECONDS).utc
|
|
48
|
+
end
|
|
49
|
+
|
|
50
|
+
# Add +count+ shed events to (error_log_id, bucket_at), creating the row if
|
|
51
|
+
# it is not there yet.
|
|
52
|
+
#
|
|
53
|
+
# Adapter-portable by hand rather than via upsert_all: the increment has to
|
|
54
|
+
# read the existing value, and the ON CONFLICT / ON DUPLICATE KEY syntaxes
|
|
55
|
+
# differ. The UPDATE-first shape means the common case (a storm flushing
|
|
56
|
+
# repeatedly into the same hour) is a single statement, and the INSERT race
|
|
57
|
+
# is resolved by retrying the UPDATE once.
|
|
58
|
+
#
|
|
59
|
+
# @return [Symbol] :written when the bucket landed, :unavailable when the
|
|
60
|
+
# rollup is permanently unusable (no table, bad args, an adapter that
|
|
61
|
+
# refuses the statement). A TRANSIENT failure raises instead.
|
|
62
|
+
#
|
|
63
|
+
# Three states, not a boolean: the caller has to tell "the bucket is
|
|
64
|
+
# permanently unavailable, degrade" from "the write failed, retry the
|
|
65
|
+
# batch". Collapsing both into false made FlushStormCounts abort the
|
|
66
|
+
# whole transaction on a host that had simply never migrated the rollup
|
|
67
|
+
# table -- rolling back the authoritative lifetime count with it.
|
|
68
|
+
def self.accumulate(error_log_id:, bucket_at:, count:)
|
|
69
|
+
return :unavailable unless error_log_id && count.to_i.positive?
|
|
70
|
+
return :unavailable unless table_exists?
|
|
71
|
+
|
|
72
|
+
bucket = bucket_for(bucket_at)
|
|
73
|
+
updated = where(error_log_id: error_log_id, bucket_at: bucket)
|
|
74
|
+
.update_all([ "count = count + ?, updated_at = ?", count.to_i, Time.current ])
|
|
75
|
+
return :written if updated.positive?
|
|
76
|
+
|
|
77
|
+
begin
|
|
78
|
+
# requires_new: a failed INSERT aborts its transaction on PostgreSQL,
|
|
79
|
+
# which would take the caller's surrounding transaction with it.
|
|
80
|
+
transaction(requires_new: true) do
|
|
81
|
+
create!(error_log_id: error_log_id, bucket_at: bucket, count: count.to_i)
|
|
82
|
+
end
|
|
83
|
+
:written
|
|
84
|
+
rescue ActiveRecord::RecordNotUnique
|
|
85
|
+
# Another process created the same bucket between the UPDATE and the
|
|
86
|
+
# INSERT. The row exists now, so the UPDATE that missed a moment ago
|
|
87
|
+
# succeeds.
|
|
88
|
+
where(error_log_id: error_log_id, bucket_at: bucket)
|
|
89
|
+
.update_all([ "count = count + ?, updated_at = ?", count.to_i, Time.current ])
|
|
90
|
+
.positive? ? :written : :unavailable
|
|
91
|
+
end
|
|
92
|
+
rescue *Commands::LogError::RETRYABLE_STORE_ERRORS => e
|
|
93
|
+
# TRANSIENT: the store may be back in a moment. Do NOT swallow it.
|
|
94
|
+
#
|
|
95
|
+
# Returning false here let FlushStormCounts finalize its batch ledger
|
|
96
|
+
# with the bucket missing, so the replay was suppressed as
|
|
97
|
+
# already-applied and those events were erased from every time window --
|
|
98
|
+
# permanently -- while the lifetime count stayed correct. The caller
|
|
99
|
+
# turns this into an abort-and-retry of the whole batch.
|
|
100
|
+
RailsErrorDashboard::Logger.debug(
|
|
101
|
+
"[RailsErrorDashboard] EventCount.accumulate hit a transient failure: #{e.class} - #{e.message}"
|
|
102
|
+
)
|
|
103
|
+
raise
|
|
104
|
+
rescue StandardError => e
|
|
105
|
+
# PERMANENT: a malformed row, a missing table, an adapter that refuses
|
|
106
|
+
# this statement. Retrying cannot help, and failing the flush would turn
|
|
107
|
+
# a lost time bucket into a lost COUNT -- the authoritative total is
|
|
108
|
+
# occurrence_count on the group, and it is already written.
|
|
109
|
+
#
|
|
110
|
+
# This is the only case the old comment actually described.
|
|
111
|
+
RailsErrorDashboard::Logger.debug(
|
|
112
|
+
"[RailsErrorDashboard] EventCount.accumulate failed permanently: #{e.class} - #{e.message}"
|
|
113
|
+
)
|
|
114
|
+
:unavailable
|
|
115
|
+
end
|
|
116
|
+
|
|
117
|
+
# Whether the rollup table is usable.
|
|
118
|
+
#
|
|
119
|
+
# A transient connection failure here is NOT "the table does not exist":
|
|
120
|
+
# swallowing it returned false before any write was attempted, so a storm
|
|
121
|
+
# flush reported success with no bucket written, and EventVolume silently
|
|
122
|
+
# dropped the bucket term from its reads. Let the transient class through
|
|
123
|
+
# so the caller can retry; only a genuinely absent table returns false.
|
|
124
|
+
def self.table_exists?
|
|
125
|
+
connection.table_exists?(table_name)
|
|
126
|
+
rescue *Commands::LogError::RETRYABLE_STORE_ERRORS
|
|
127
|
+
raise
|
|
128
|
+
rescue StandardError
|
|
129
|
+
false
|
|
130
|
+
end
|
|
131
|
+
end
|
|
132
|
+
end
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RailsErrorDashboard
|
|
4
|
+
# An interval whose per-event timing evidence was lost.
|
|
5
|
+
#
|
|
6
|
+
# See the migration for why this is its own table rather than a flag on the
|
|
7
|
+
# storm episode: the episode is optional, and it is written after the counts
|
|
8
|
+
# transaction has already committed.
|
|
9
|
+
#
|
|
10
|
+
# Inherits ErrorLogsRecord so separate-database routing applies.
|
|
11
|
+
class EventTimingGap < ErrorLogsRecord
|
|
12
|
+
self.table_name = "rails_error_dashboard_event_timing_gaps"
|
|
13
|
+
|
|
14
|
+
# Gaps overlapping [from, to). Open-ended when +to+ is nil.
|
|
15
|
+
#
|
|
16
|
+
# Overlap, not containment: a gap that began before the window and runs
|
|
17
|
+
# into it still makes that window's timing unreliable. Checking only
|
|
18
|
+
# "starts inside the window" is the mistake the episode predicate made.
|
|
19
|
+
scope :overlapping, ->(from, to = nil) {
|
|
20
|
+
scope = where(arel_table[:covered_until].gteq(from))
|
|
21
|
+
to ? scope.where(arel_table[:covered_from].lt(to)) : scope
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
# Whether any recorded gap affects the given window.
|
|
25
|
+
#
|
|
26
|
+
# Every guard is deliberate: the table may not be migrated on an older
|
|
27
|
+
# host, and this is read on the dashboard path, which must never raise.
|
|
28
|
+
#
|
|
29
|
+
# @param from [Time] start of the window being displayed
|
|
30
|
+
# @param application_id [Integer, nil]
|
|
31
|
+
# @return [Boolean]
|
|
32
|
+
def self.affecting?(from, application_id: nil)
|
|
33
|
+
return false unless table_exists?
|
|
34
|
+
|
|
35
|
+
scope = overlapping(from)
|
|
36
|
+
# A NULL application_id means the gap applies everywhere, so it is
|
|
37
|
+
# included whatever is being filtered for.
|
|
38
|
+
scope = scope.where(application_id: [ application_id, nil ]) if application_id.present?
|
|
39
|
+
scope.exists?
|
|
40
|
+
rescue StandardError => e
|
|
41
|
+
RailsErrorDashboard::Logger.debug(
|
|
42
|
+
"[RailsErrorDashboard] EventTimingGap.affecting? failed: #{e.class} - #{e.message}"
|
|
43
|
+
)
|
|
44
|
+
false
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
def self.table_exists?
|
|
48
|
+
connection.table_exists?(table_name)
|
|
49
|
+
rescue *Commands::LogError::RETRYABLE_STORE_ERRORS
|
|
50
|
+
raise
|
|
51
|
+
rescue StandardError
|
|
52
|
+
false
|
|
53
|
+
end
|
|
54
|
+
end
|
|
55
|
+
end
|