rails_error_dashboard 0.12.1 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/app/controllers/rails_error_dashboard/application_controller.rb +92 -11
- data/app/controllers/rails_error_dashboard/errors_controller.rb +111 -34
- data/app/controllers/rails_error_dashboard/webhooks_controller.rb +3 -0
- data/app/helpers/rails_error_dashboard/application_helper.rb +46 -1
- data/app/helpers/rails_error_dashboard/backtrace_helper.rb +8 -0
- data/app/jobs/rails_error_dashboard/async_error_logging_job.rb +10 -0
- data/app/jobs/rails_error_dashboard/concerns/plain_channel_message.rb +71 -0
- data/app/jobs/rails_error_dashboard/notification_burst_summary_job.rb +55 -0
- data/app/jobs/rails_error_dashboard/retention_cleanup_job.rb +122 -1
- data/app/jobs/rails_error_dashboard/storm_flush_job.rb +7 -4
- data/app/jobs/rails_error_dashboard/storm_notification_job.rb +8 -44
- data/app/models/rails_error_dashboard/error_baseline.rb +12 -9
- data/app/models/rails_error_dashboard/error_comment.rb +0 -5
- data/app/models/rails_error_dashboard/error_log.rb +19 -3
- data/app/models/rails_error_dashboard/error_logs_record.rb +34 -0
- data/app/models/rails_error_dashboard/error_occurrence.rb +9 -1
- data/app/models/rails_error_dashboard/event_count.rb +132 -0
- data/app/models/rails_error_dashboard/event_timing_gap.rb +55 -0
- data/app/views/layouts/rails_error_dashboard.html.erb +58 -5
- data/app/views/rails_error_dashboard/errors/_discussion.html.erb +18 -11
- data/app/views/rails_error_dashboard/errors/_issue_section.html.erb +22 -8
- data/app/views/rails_error_dashboard/errors/_request_context.html.erb +2 -0
- data/app/views/rails_error_dashboard/errors/_sidebar_metadata.html.erb +15 -6
- data/app/views/rails_error_dashboard/errors/analytics.html.erb +8 -8
- data/app/views/rails_error_dashboard/errors/diagnostic_dumps.html.erb +1 -1
- data/app/views/rails_error_dashboard/errors/index.html.erb +6 -1
- data/app/views/rails_error_dashboard/errors/overview.html.erb +12 -0
- data/app/views/rails_error_dashboard/errors/platform_comparison.html.erb +7 -7
- data/app/views/rails_error_dashboard/errors/releases.html.erb +2 -2
- data/app/views/rails_error_dashboard/errors/settings.html.erb +2 -0
- data/app/views/rails_error_dashboard/errors/show.html.erb +3 -3
- data/config/locales/de.yml +30 -0
- data/config/locales/en.yml +37 -0
- data/config/locales/es.yml +30 -0
- data/config/locales/fr.yml +30 -0
- data/config/locales/it.yml +30 -0
- data/config/locales/ja.yml +30 -0
- data/config/locales/pl.yml +30 -0
- data/config/locales/pt-BR.yml +30 -0
- data/config/locales/ru.yml +30 -0
- data/config/locales/uk.yml +30 -0
- data/config/locales/zh-CN.yml +30 -0
- data/db/migrate/20260917000001_add_last_notified_at_to_error_logs.rb +40 -0
- data/db/migrate/20260919000001_create_event_counts.rb +71 -0
- data/db/migrate/20260920000001_add_buckets_incomplete_to_storm_events.rb +25 -0
- data/db/migrate/20260920000002_create_event_timing_gaps.rb +55 -0
- data/lib/generators/rails_error_dashboard/install/templates/initializer.rb +12 -1
- data/lib/rails_error_dashboard/commands/assign_error.rb +12 -2
- data/lib/rails_error_dashboard/commands/backfill_environments.rb +2 -0
- data/lib/rails_error_dashboard/commands/backfill_resolved_at.rb +42 -0
- data/lib/rails_error_dashboard/commands/batch_delete_errors.rb +1 -0
- data/lib/rails_error_dashboard/commands/batch_mute_errors.rb +2 -0
- data/lib/rails_error_dashboard/commands/batch_resolve_errors.rb +2 -0
- data/lib/rails_error_dashboard/commands/batch_unmute_errors.rb +2 -0
- data/lib/rails_error_dashboard/commands/find_or_increment_error.rb +109 -11
- data/lib/rails_error_dashboard/commands/flush_rack_attack_events.rb +3 -1
- data/lib/rails_error_dashboard/commands/flush_storm_counts.rb +252 -12
- data/lib/rails_error_dashboard/commands/flush_swallowed_exceptions.rb +5 -2
- data/lib/rails_error_dashboard/commands/link_existing_issue.rb +1 -0
- data/lib/rails_error_dashboard/commands/log_error.rb +291 -40
- data/lib/rails_error_dashboard/commands/mute_error.rb +1 -0
- data/lib/rails_error_dashboard/commands/resolve_error.rb +2 -0
- data/lib/rails_error_dashboard/commands/scrub_invalid_encoding.rb +104 -0
- data/lib/rails_error_dashboard/commands/snooze_error.rb +34 -10
- data/lib/rails_error_dashboard/commands/unmute_error.rb +1 -0
- data/lib/rails_error_dashboard/commands/update_error_priority.rb +23 -2
- data/lib/rails_error_dashboard/commands/update_error_status.rb +33 -6
- data/lib/rails_error_dashboard/configuration.rb +39 -1
- data/lib/rails_error_dashboard/engine.rb +28 -0
- data/lib/rails_error_dashboard/manual_error_reporter.rb +16 -5
- data/lib/rails_error_dashboard/queries/analytics_stats.rb +89 -30
- data/lib/rails_error_dashboard/queries/baseline_stats.rb +107 -0
- data/lib/rails_error_dashboard/queries/dashboard_stats.rb +216 -74
- data/lib/rails_error_dashboard/queries/error_correlation.rb +6 -3
- data/lib/rails_error_dashboard/queries/errors_list.rb +15 -2
- data/lib/rails_error_dashboard/queries/event_volume.rb +503 -0
- data/lib/rails_error_dashboard/queries/similar_errors.rb +1 -1
- data/lib/rails_error_dashboard/services/analytics_cache_manager.rb +43 -18
- data/lib/rails_error_dashboard/services/backtrace_processor.rb +3 -1
- data/lib/rails_error_dashboard/services/breadcrumb_collector.rb +23 -0
- data/lib/rails_error_dashboard/services/cause_chain_extractor.rb +3 -1
- data/lib/rails_error_dashboard/services/codeberg_issue_client.rb +13 -2
- data/lib/rails_error_dashboard/services/diagnostic_dump_generator.rb +5 -3
- data/lib/rails_error_dashboard/services/encoding_sanitizer.rb +80 -0
- data/lib/rails_error_dashboard/services/error_broadcaster.rb +180 -32
- data/lib/rails_error_dashboard/services/error_hash_generator.rb +9 -3
- data/lib/rails_error_dashboard/services/error_notification_dispatcher.rb +13 -0
- data/lib/rails_error_dashboard/services/exception_filter.rb +55 -0
- data/lib/rails_error_dashboard/services/git_head_reader.rb +102 -0
- data/lib/rails_error_dashboard/services/notification_throttler.rb +172 -28
- data/lib/rails_error_dashboard/services/sensitive_data_filter.rb +47 -1
- data/lib/rails_error_dashboard/services/storm_protection/circuit_breaker.rb +59 -3
- data/lib/rails_error_dashboard/services/storm_protection/count_buffer.rb +64 -6
- data/lib/rails_error_dashboard/services/storm_protection/fingerprint_buckets.rb +25 -2
- data/lib/rails_error_dashboard/services/storm_protection/gate.rb +32 -5
- data/lib/rails_error_dashboard/services/swallowed_exception_tracker.rb +99 -23
- data/lib/rails_error_dashboard/services/url_safety.rb +40 -0
- data/lib/rails_error_dashboard/services/variable_serializer.rb +125 -10
- data/lib/rails_error_dashboard/subscribers/breadcrumb_subscriber.rb +111 -0
- data/lib/rails_error_dashboard/subscribers/issue_tracker_subscriber.rb +11 -2
- data/lib/rails_error_dashboard/value_objects/error_context.rb +40 -3
- data/lib/rails_error_dashboard/version.rb +1 -1
- data/lib/rails_error_dashboard.rb +34 -0
- data/lib/tasks/error_dashboard.rake +54 -4
- metadata +16 -2
|
@@ -10,8 +10,9 @@ module RailsErrorDashboard
|
|
|
10
10
|
# and two concurrent captures could both read count N and both write N+1.
|
|
11
11
|
#
|
|
12
12
|
# Search order:
|
|
13
|
+
# 0. wont_fix errors with same hash (any age) → increment, status untouched
|
|
13
14
|
# 1. Unresolved errors with same hash within 24 hours → increment occurrence count
|
|
14
|
-
# 2. Resolved
|
|
15
|
+
# 2. Resolved errors with same hash (any age) → reopen and increment
|
|
15
16
|
# 3. No match → create new error record
|
|
16
17
|
#
|
|
17
18
|
# Environment is a MATCH dimension, not part of the hash: the same error in
|
|
@@ -42,11 +43,15 @@ module RailsErrorDashboard
|
|
|
42
43
|
|
|
43
44
|
def call
|
|
44
45
|
ErrorLog.transaction do
|
|
46
|
+
# Priority 0: a wont_fix row absorbs its recurrences, at any age
|
|
47
|
+
sticky = find_wont_fix
|
|
48
|
+
next increment_existing(sticky) if sticky
|
|
49
|
+
|
|
45
50
|
# Priority 1: Find unresolved match (existing behavior)
|
|
46
51
|
existing = find_unresolved
|
|
47
52
|
next increment_existing(existing) if existing
|
|
48
53
|
|
|
49
|
-
# Priority 2: Find resolved
|
|
54
|
+
# Priority 2: Find resolved match → reopen
|
|
50
55
|
resolved = find_resolved
|
|
51
56
|
next reopen_existing(resolved) if resolved
|
|
52
57
|
|
|
@@ -57,9 +62,26 @@ module RailsErrorDashboard
|
|
|
57
62
|
|
|
58
63
|
private
|
|
59
64
|
|
|
65
|
+
# The three lookups are DISJOINT by status, so which row a recurrence
|
|
66
|
+
# lands on never depends on the order they happen to run in.
|
|
67
|
+
|
|
68
|
+
# "Won't fix" is a decision that the error recurs and will not be acted
|
|
69
|
+
# on, so it has no time window: the row counts its recurrences for as
|
|
70
|
+
# long as it keeps the status. It used to be matched by find_unresolved
|
|
71
|
+
# for 24 hours and then REOPENED by find_resolved -- sticky for a day,
|
|
72
|
+
# after which the triage decision was silently thrown away.
|
|
73
|
+
def find_wont_fix
|
|
74
|
+
with_environment(
|
|
75
|
+
ErrorLog
|
|
76
|
+
.where(error_hash: @error_hash)
|
|
77
|
+
.where(application_id: @attributes[:application_id])
|
|
78
|
+
.where(status: "wont_fix")
|
|
79
|
+
).lock.order(last_seen_at: :desc).first
|
|
80
|
+
end
|
|
81
|
+
|
|
60
82
|
def find_unresolved
|
|
61
83
|
with_environment(
|
|
62
|
-
ErrorLog.unresolved
|
|
84
|
+
not_wont_fix(ErrorLog.unresolved)
|
|
63
85
|
.where(error_hash: @error_hash)
|
|
64
86
|
.where(application_id: @attributes[:application_id])
|
|
65
87
|
.where("occurred_at >= ?", 24.hours.ago)
|
|
@@ -71,10 +93,16 @@ module RailsErrorDashboard
|
|
|
71
93
|
ErrorLog
|
|
72
94
|
.where(error_hash: @error_hash)
|
|
73
95
|
.where(application_id: @attributes[:application_id])
|
|
74
|
-
.where(status:
|
|
96
|
+
.where(status: "resolved")
|
|
75
97
|
).lock.order(last_seen_at: :desc).first
|
|
76
98
|
end
|
|
77
99
|
|
|
100
|
+
# Spelled out rather than where.not(status: "wont_fix"): in SQL that also
|
|
101
|
+
# drops every row whose status is NULL.
|
|
102
|
+
def not_wont_fix(scope)
|
|
103
|
+
scope.where("status IS NULL OR status <> ?", "wont_fix")
|
|
104
|
+
end
|
|
105
|
+
|
|
78
106
|
# Restrict to this occurrence's environment or a legacy NULL row, exact
|
|
79
107
|
# first. Literal SQL, no interpolation. A blank environment (column not
|
|
80
108
|
# migrated yet, or an attribute-less caller) leaves the scope unchanged.
|
|
@@ -109,21 +137,78 @@ module RailsErrorDashboard
|
|
|
109
137
|
# leaves both the snapshot and its provenance alone -- otherwise the row
|
|
110
138
|
# would claim a fresh capture time for evidence from an older event,
|
|
111
139
|
# which is precisely the confusion this exists to remove.
|
|
112
|
-
def context_provenance(refreshed)
|
|
140
|
+
def context_provenance(refreshed, error = nil)
|
|
113
141
|
return {} unless refreshed.any? || refreshed_request_identity?
|
|
114
142
|
return {} unless ErrorLog.column_names.include?("context_captured_at")
|
|
115
143
|
|
|
116
144
|
provenance = { context_captured_at: @attributes[:occurred_at] || Time.current }
|
|
117
145
|
if ErrorLog.column_names.include?("context_fidelity")
|
|
118
|
-
provenance[:context_fidelity] =
|
|
146
|
+
provenance[:context_fidelity] =
|
|
147
|
+
@attributes[:_context_fidelity].presence || snapshot_fidelity(error)
|
|
119
148
|
end
|
|
120
149
|
provenance
|
|
121
150
|
end
|
|
122
151
|
|
|
152
|
+
# "full" means every displayed field came from THIS occurrence. When the
|
|
153
|
+
# `||` chain keeps an older value beside a newly refreshed one, the row
|
|
154
|
+
# is showing a mixture of two events, and calling that a fresh full
|
|
155
|
+
# capture is what made the snapshot unreadable: a new request URL sat
|
|
156
|
+
# beside a previous occurrence's user and locals under one timestamp.
|
|
157
|
+
#
|
|
158
|
+
# Keeping the older value is still the right behaviour -- a useful
|
|
159
|
+
# exemplar beats a blank one -- so only the LABEL changes.
|
|
160
|
+
def snapshot_fidelity(error)
|
|
161
|
+
return "full" if error.nil?
|
|
162
|
+
|
|
163
|
+
retains_older_value?(error) ? "partial" : "full"
|
|
164
|
+
end
|
|
165
|
+
|
|
166
|
+
# Every field whose stored value this capture could RETAIN from an
|
|
167
|
+
# earlier occurrence. Defined once, so a field cannot join the displayed
|
|
168
|
+
# snapshot without joining the provenance policy -- which is exactly how
|
|
169
|
+
# local variables came to be shown beside a "full" label and a newer
|
|
170
|
+
# timestamp while belonging to a different event.
|
|
171
|
+
#
|
|
172
|
+
# Request identity plus the context payloads: both are subject to the
|
|
173
|
+
# same `||` chain, and a reader cannot tell them apart on the page.
|
|
174
|
+
PROVENANCE_TRACKED = (REFRESHED_REQUEST_IDENTITY + REFRESHED_CONTEXT).uniq.freeze
|
|
175
|
+
|
|
176
|
+
# True when the row already holds a displayed value that this occurrence
|
|
177
|
+
# did NOT supply, so the `||` chain is about to keep it and the stored
|
|
178
|
+
# snapshot will describe two different events.
|
|
179
|
+
def retains_older_value?(error)
|
|
180
|
+
PROVENANCE_TRACKED.any? do |key|
|
|
181
|
+
next false unless ErrorLog.column_names.include?(key.to_s)
|
|
182
|
+
|
|
183
|
+
@attributes[key].nil? && previous_value(error, key).present?
|
|
184
|
+
end
|
|
185
|
+
end
|
|
186
|
+
|
|
187
|
+
# read_attribute, never public_send: :instance_variables would otherwise
|
|
188
|
+
# resolve to Ruby's own Object#instance_variables if the generated
|
|
189
|
+
# attribute method were ever absent (ignored_columns, load order), and
|
|
190
|
+
# silently compare an Array of symbols against captured context.
|
|
191
|
+
def previous_value(error, key)
|
|
192
|
+
error.read_attribute(key)
|
|
193
|
+
rescue StandardError
|
|
194
|
+
nil
|
|
195
|
+
end
|
|
196
|
+
|
|
123
197
|
def refreshed_request_identity?
|
|
124
198
|
REFRESHED_REQUEST_IDENTITY.any? { |key| !@attributes[key].nil? }
|
|
125
199
|
end
|
|
126
200
|
|
|
201
|
+
# Inside the 24-hour matching window find_unresolved uses, so the group
|
|
202
|
+
# this creates can still be found by its own recurrences.
|
|
203
|
+
def clamped_group_time(time)
|
|
204
|
+
return Time.current if time.blank?
|
|
205
|
+
|
|
206
|
+
floor = 24.hours.ago + 1.minute
|
|
207
|
+
time < floor ? floor : time
|
|
208
|
+
rescue StandardError
|
|
209
|
+
Time.current
|
|
210
|
+
end
|
|
211
|
+
|
|
127
212
|
# A group first seen during a storm has a MINIMAL exemplar: the flush job
|
|
128
213
|
# could only record the first app frame, because a counted-only event
|
|
129
214
|
# captures no backtrace. The comment there promises the next occurrence
|
|
@@ -177,7 +262,7 @@ module RailsErrorDashboard
|
|
|
177
262
|
user_agent: @attributes[:user_agent] || error.user_agent,
|
|
178
263
|
ip_address: @attributes[:ip_address] || error.ip_address,
|
|
179
264
|
**refreshed,
|
|
180
|
-
**context_provenance(refreshed),
|
|
265
|
+
**context_provenance(refreshed, error),
|
|
181
266
|
**backtrace_upgrade(error),
|
|
182
267
|
**environment_adoption(error)
|
|
183
268
|
)
|
|
@@ -197,7 +282,7 @@ module RailsErrorDashboard
|
|
|
197
282
|
user_agent: @attributes[:user_agent] || error.user_agent,
|
|
198
283
|
ip_address: @attributes[:ip_address] || error.ip_address,
|
|
199
284
|
**(refreshed = latest_context),
|
|
200
|
-
**context_provenance(refreshed),
|
|
285
|
+
**context_provenance(refreshed, error),
|
|
201
286
|
**backtrace_upgrade(error),
|
|
202
287
|
**environment_adoption(error)
|
|
203
288
|
}
|
|
@@ -217,6 +302,15 @@ module RailsErrorDashboard
|
|
|
217
302
|
attrs = @attributes.reject { |key, _| key.to_s.start_with?("_") }
|
|
218
303
|
attrs = attrs.reverse_merge(resolved: false)
|
|
219
304
|
|
|
305
|
+
# The GROUP's occurred_at is clamped into the matching window, even
|
|
306
|
+
# when the EVENT is older. find_unresolved matches on
|
|
307
|
+
# `occurred_at >= 24.hours.ago`, so a backdated report (a mobile client
|
|
308
|
+
# flushing a queue it collected offline) would otherwise create a row
|
|
309
|
+
# that can never be matched again -- every recurrence making yet
|
|
310
|
+
# another group. The event's true time is preserved on its occurrence
|
|
311
|
+
# row, which is what the time-window queries read.
|
|
312
|
+
attrs[:occurred_at] = clamped_group_time(attrs[:occurred_at])
|
|
313
|
+
|
|
220
314
|
if ErrorLog.column_names.include?("context_captured_at")
|
|
221
315
|
attrs[:context_captured_at] ||= @attributes[:occurred_at] || Time.current
|
|
222
316
|
end
|
|
@@ -235,9 +329,13 @@ module RailsErrorDashboard
|
|
|
235
329
|
ErrorLog.create!(new_record_attributes)
|
|
236
330
|
end
|
|
237
331
|
rescue ActiveRecord::RecordNotUnique
|
|
238
|
-
# Race condition: another process created the same error
|
|
332
|
+
# Race condition: another process created the same error. Same three
|
|
333
|
+
# lookups, same order, as the first pass.
|
|
334
|
+
retry_sticky = find_wont_fix
|
|
335
|
+
return increment_existing(retry_sticky) if retry_sticky
|
|
336
|
+
|
|
239
337
|
retry_existing = with_environment(
|
|
240
|
-
ErrorLog.unresolved
|
|
338
|
+
not_wont_fix(ErrorLog.unresolved)
|
|
241
339
|
.where(error_hash: @error_hash)
|
|
242
340
|
.where(application_id: @attributes[:application_id])
|
|
243
341
|
.where("occurred_at >= ?", 24.hours.ago)
|
|
@@ -257,7 +355,7 @@ module RailsErrorDashboard
|
|
|
257
355
|
ErrorLog
|
|
258
356
|
.where(error_hash: @error_hash)
|
|
259
357
|
.where(application_id: @attributes[:application_id])
|
|
260
|
-
.where(status:
|
|
358
|
+
.where(status: "resolved")
|
|
261
359
|
).lock.first
|
|
262
360
|
|
|
263
361
|
if retry_resolved
|
|
@@ -28,8 +28,10 @@ module RailsErrorDashboard
|
|
|
28
28
|
app_id = current_application_id
|
|
29
29
|
|
|
30
30
|
@counts.each do |key, count|
|
|
31
|
+
# Path and user agent are attacker-supplied, so invalid bytes are the
|
|
32
|
+
# expected case here. Scrub the whole key before it is split.
|
|
31
33
|
rule, match_type, discriminator, path, http_method, user_agent =
|
|
32
|
-
Services::RackAttackTracker.parse_key(key)
|
|
34
|
+
Services::RackAttackTracker.parse_key(Services::EncodingSanitizer.scrub(key.to_s))
|
|
33
35
|
|
|
34
36
|
next if rule.blank? || match_type.blank?
|
|
35
37
|
|
|
@@ -16,13 +16,26 @@ module RailsErrorDashboard
|
|
|
16
16
|
# Counts are exact. Notifications are NOT dispatched from here — during a
|
|
17
17
|
# storm they're suppressed by design; the storm notification covers it.
|
|
18
18
|
class FlushStormCounts
|
|
19
|
+
# A bucket write failed in a way that may succeed on retry. Raised so the
|
|
20
|
+
# per-entry rescue in #call classifies it exactly as it classifies a
|
|
21
|
+
# transient COUNT failure: roll the batch back, leave the ledger
|
|
22
|
+
# unclaimed, let the job retry the whole batch intact.
|
|
23
|
+
class EventCountWriteFailed < StandardError; end
|
|
24
|
+
|
|
19
25
|
def self.call(entries:, overflow: 0, episode: nil, batch_id: nil)
|
|
20
26
|
new(entries: entries, overflow: overflow, episode: episode, batch_id: batch_id).call
|
|
21
27
|
end
|
|
22
28
|
|
|
23
29
|
def initialize(entries:, overflow: 0, episode: nil, batch_id: nil)
|
|
24
|
-
|
|
30
|
+
# The gate scrubs what it buffers, so this is normally a no-op scan. It
|
|
31
|
+
# covers entries buffered by an older release and direct callers: the
|
|
32
|
+
# exemplar becomes an ErrorLog row, and its message is matched by regex.
|
|
33
|
+
@entries = Array(Services::EncodingSanitizer.scrub_deep(entries))
|
|
25
34
|
@overflow = overflow.to_i
|
|
35
|
+
# Set when a bucket write was permanently unavailable. The counts are
|
|
36
|
+
# still correct; only their placement in TIME is missing, and a caller
|
|
37
|
+
# reading a time window deserves to know that.
|
|
38
|
+
@buckets_incomplete = false
|
|
26
39
|
@episode = episode
|
|
27
40
|
@batch_id = batch_id
|
|
28
41
|
end
|
|
@@ -31,6 +44,7 @@ module RailsErrorDashboard
|
|
|
31
44
|
application = resolve_application
|
|
32
45
|
counted = 0
|
|
33
46
|
failed = 0
|
|
47
|
+
aborted = false
|
|
34
48
|
|
|
35
49
|
# Counts are applied additively (occurrence_count + N), which is not
|
|
36
50
|
# idempotent: delivering the same snapshot twice counted it twice. The
|
|
@@ -52,6 +66,20 @@ module RailsErrorDashboard
|
|
|
52
66
|
@entries.each do |entry|
|
|
53
67
|
entry = entry.with_indifferent_access if entry.respond_to?(:with_indifferent_access)
|
|
54
68
|
counted += reconcile_entry(entry, application)
|
|
69
|
+
rescue EventCountWriteFailed, *Commands::LogError::RETRYABLE_STORE_ERRORS => e
|
|
70
|
+
# A transient store failure is NOT a bad entry. Claiming the batch
|
|
71
|
+
# here would commit the ledger row and strand every entry not yet
|
|
72
|
+
# applied: the retry is then suppressed as a replay and those events
|
|
73
|
+
# are lost for good. Re-raise so the whole transaction rolls back --
|
|
74
|
+
# nothing was committed, so nothing can double -- and let the job
|
|
75
|
+
# retry the batch intact. The generic rescue below still keeps a
|
|
76
|
+
# permanently malformed entry from poisoning its batch.
|
|
77
|
+
failed += 1
|
|
78
|
+
aborted = true
|
|
79
|
+
RailsErrorDashboard::Logger.error(
|
|
80
|
+
"[RailsErrorDashboard] Storm batch aborted by a transient store failure: #{e.class} - #{e.message}"
|
|
81
|
+
)
|
|
82
|
+
raise
|
|
55
83
|
rescue => e
|
|
56
84
|
# A corrupt (non-Hash) entry must not abort the whole batch — and the
|
|
57
85
|
# log line itself must not assume `entry` is subscriptable (an Integer
|
|
@@ -67,26 +95,56 @@ module RailsErrorDashboard
|
|
|
67
95
|
# roll the claim back and let the job retry the whole batch.
|
|
68
96
|
raise ActiveRecord::Rollback if failed.positive? && counted.zero?
|
|
69
97
|
|
|
98
|
+
# INSIDE the transaction, deliberately.
|
|
99
|
+
#
|
|
100
|
+
# The counts and the record that their timing is unreliable have to
|
|
101
|
+
# land together or not at all. Writing this after the commit (as the
|
|
102
|
+
# storm-episode marker did) meant a transient failure lost the
|
|
103
|
+
# marker while the ledger had already recorded the batch as applied
|
|
104
|
+
# -- the replay was then suppressed and the gap was never recorded,
|
|
105
|
+
# so the dashboard reported completeness it could not vouch for.
|
|
106
|
+
#
|
|
107
|
+
# If this write fails, the whole batch rolls back and stays
|
|
108
|
+
# replayable. Counts whose unreliability we cannot record are worth
|
|
109
|
+
# retrying, not committing silently.
|
|
110
|
+
record_timing_gap!(counted) if @buckets_incomplete && counted.positive?
|
|
111
|
+
|
|
70
112
|
finalize_batch!(ledger, counted)
|
|
71
113
|
end
|
|
72
114
|
|
|
73
115
|
# Every entry failed and none was written. Reporting success with
|
|
74
116
|
# reconciled: 0 made a total loss indistinguishable from an empty
|
|
75
117
|
# batch, so the job acknowledged counts that never reached the
|
|
76
|
-
# database.
|
|
77
|
-
#
|
|
118
|
+
# database.
|
|
119
|
+
#
|
|
120
|
+
# Reaching here means every failure was PERMANENT -- a transient store
|
|
121
|
+
# failure re-raises above and rolls the whole batch back. Partial
|
|
122
|
+
# success over permanent failures stays successful: the entries that
|
|
123
|
+
# were written are written, replaying would double them, and retrying a
|
|
124
|
+
# corrupt payload only loops forever.
|
|
78
125
|
if failed.positive? && counted.zero?
|
|
79
|
-
return { success: false, reconciled: 0, failed: failed, overflow: @overflow,
|
|
126
|
+
return { success: false, retryable: false, reconciled: 0, failed: failed, overflow: @overflow,
|
|
127
|
+
buckets_incomplete: @buckets_incomplete,
|
|
80
128
|
error: "all #{failed} entries failed to reconcile" }
|
|
81
129
|
end
|
|
82
130
|
|
|
83
131
|
upsert_storm_event(counted)
|
|
84
|
-
{ success: true, reconciled: counted, failed: failed, overflow: @overflow }
|
|
132
|
+
result = { success: true, reconciled: counted, failed: failed, overflow: @overflow }
|
|
133
|
+
result[:buckets_incomplete] = true if @buckets_incomplete
|
|
134
|
+
result
|
|
85
135
|
rescue => e
|
|
86
136
|
RailsErrorDashboard::Logger.error(
|
|
87
137
|
"[RailsErrorDashboard] FlushStormCounts failed: #{e.class} - #{e.message}"
|
|
88
138
|
)
|
|
89
|
-
|
|
139
|
+
# retryable: true says "the batch is intact, replay it" -- nothing was
|
|
140
|
+
# committed, so the job can retry without doubling. A permanent failure
|
|
141
|
+
# carries no such promise.
|
|
142
|
+
retryable = Commands::LogError::RETRYABLE_STORE_ERRORS.any? { |klass| e.is_a?(klass) }
|
|
143
|
+
result = { success: false, retryable: retryable, error: "#{e.class}: #{e.message}" }
|
|
144
|
+
# reconciled: 0 because the transaction rolled back -- whatever this
|
|
145
|
+
# batch had counted in memory never reached the database.
|
|
146
|
+
result.merge!(reconciled: 0, failed: failed, overflow: @overflow) if aborted
|
|
147
|
+
result
|
|
90
148
|
end
|
|
91
149
|
|
|
92
150
|
private
|
|
@@ -145,7 +203,68 @@ module RailsErrorDashboard
|
|
|
145
203
|
end
|
|
146
204
|
|
|
147
205
|
|
|
206
|
+
# Reconcile one buffered entry and, on every path that adds counts, give
|
|
207
|
+
# those shed events a TIME BUCKET as well as a total.
|
|
208
|
+
#
|
|
209
|
+
# occurrence_count alone is a lifetime counter: it says how many events
|
|
210
|
+
# there were but not when, so no window query can place them. Ordinary
|
|
211
|
+
# captures carry their own ErrorOccurrence row; shed events write none by
|
|
212
|
+
# design, which is what made them invisible to "errors today". The bucket
|
|
213
|
+
# written here is what Queries::EventVolume adds to the occurrence rows.
|
|
148
214
|
def reconcile_entry(entry, application)
|
|
215
|
+
error_log_id = nil
|
|
216
|
+
count = reconcile_entry_count(entry, application) { |id| error_log_id = id }
|
|
217
|
+
write_event_buckets(entry, error_log_id, count) if count.positive? && error_log_id
|
|
218
|
+
count
|
|
219
|
+
end
|
|
220
|
+
|
|
221
|
+
# Write one row per BUCKET the producer recorded, not one row for the
|
|
222
|
+
# whole entry.
|
|
223
|
+
#
|
|
224
|
+
# The buffer tallies events per 15-minute bucket precisely so this does
|
|
225
|
+
# not have to guess: assigning the entry's whole total to last_seen_at's
|
|
226
|
+
# bucket put an event from 23:59:59 and one from 00:00:01 on the same
|
|
227
|
+
# day. A payload from an older release carries no buckets, so it falls
|
|
228
|
+
# back to the old behaviour rather than losing the count.
|
|
229
|
+
def write_event_buckets(entry, error_log_id, count)
|
|
230
|
+
buckets = entry["buckets"]
|
|
231
|
+
buckets = nil unless buckets.is_a?(Hash) && buckets.any?
|
|
232
|
+
|
|
233
|
+
pairs =
|
|
234
|
+
if buckets
|
|
235
|
+
buckets.map { |at, n| [ Time.zone.at(at.to_i), n.to_i ] }
|
|
236
|
+
else
|
|
237
|
+
[ [ parse_time(entry["last_seen_at"]) || Time.current, count ] ]
|
|
238
|
+
end
|
|
239
|
+
|
|
240
|
+
pairs.each do |bucket_at, n|
|
|
241
|
+
next unless n.positive?
|
|
242
|
+
|
|
243
|
+
# Three outcomes, three responses.
|
|
244
|
+
#
|
|
245
|
+
# :written -- done.
|
|
246
|
+
# raises -- TRANSIENT. Handled by the caller's rescue, which
|
|
247
|
+
# aborts and replays the batch intact. Swallowing it
|
|
248
|
+
# finalized the ledger with the bucket missing, so
|
|
249
|
+
# the replay was suppressed as already-applied and
|
|
250
|
+
# the time window lost those events permanently
|
|
251
|
+
# while the lifetime count stayed correct.
|
|
252
|
+
# :unavailable -- PERMANENT. Degrade: the rollup is simply not
|
|
253
|
+
# usable on this host (never migrated, adapter
|
|
254
|
+
# refuses the statement). Raising here rolled back
|
|
255
|
+
# the surrounding transaction and destroyed the
|
|
256
|
+
# authoritative lifetime count along with it --
|
|
257
|
+
# turning a missing time bucket into a lost count,
|
|
258
|
+
# which is strictly worse. Record it instead, so the
|
|
259
|
+
# result can say the timing evidence is incomplete.
|
|
260
|
+
case EventCount.accumulate(error_log_id: error_log_id, bucket_at: bucket_at, count: n)
|
|
261
|
+
when :written then next
|
|
262
|
+
else @buckets_incomplete = true
|
|
263
|
+
end
|
|
264
|
+
end
|
|
265
|
+
end
|
|
266
|
+
|
|
267
|
+
def reconcile_entry_count(entry, application)
|
|
149
268
|
count = entry["count"].to_i
|
|
150
269
|
return 0 if count <= 0
|
|
151
270
|
|
|
@@ -165,7 +284,12 @@ module RailsErrorDashboard
|
|
|
165
284
|
# long-running unresolved error legitimately owns several unresolved
|
|
166
285
|
# rows. An update_all across the whole hash would add N to every one
|
|
167
286
|
# of them — seven real events becoming twelve counted occurrences.
|
|
168
|
-
|
|
287
|
+
#
|
|
288
|
+
# Priority 0, tried first: a wont_fix row absorbs its recurrences at any
|
|
289
|
+
# age and keeps its status (FindOrIncrementError#find_wont_fix). Same
|
|
290
|
+
# atomic increment, so it shares the branch below.
|
|
291
|
+
target = wont_fix_target(error_hash, application, env) ||
|
|
292
|
+
unresolved_target(error_hash, application, env)
|
|
169
293
|
if target
|
|
170
294
|
if env && target.environment.blank?
|
|
171
295
|
ErrorLog.where(id: target.id).update_all([
|
|
@@ -177,30 +301,56 @@ module RailsErrorDashboard
|
|
|
177
301
|
"occurrence_count = occurrence_count + ?, last_seen_at = ?", count, last_seen
|
|
178
302
|
])
|
|
179
303
|
end
|
|
304
|
+
yield target.id if block_given?
|
|
180
305
|
return count
|
|
181
306
|
end
|
|
182
307
|
|
|
183
|
-
# Priority 2: resolved
|
|
308
|
+
# Priority 2: resolved match — reopen, mirroring
|
|
184
309
|
# FindOrIncrementError so storm recurrences don't stay buried
|
|
185
310
|
resolved_scope = ErrorLog
|
|
186
311
|
.where(error_hash: error_hash, application_id: application.id)
|
|
187
|
-
.where(status:
|
|
312
|
+
.where(status: "resolved")
|
|
188
313
|
if env
|
|
189
314
|
resolved_scope = resolved_scope.where(environment: [ env, nil ])
|
|
190
315
|
.order(Arel.sql("CASE WHEN environment IS NULL THEN 1 ELSE 0 END"))
|
|
191
316
|
end
|
|
192
|
-
|
|
317
|
+
# .lock (SELECT ... FOR UPDATE) held to commit by the transaction opened
|
|
318
|
+
# in #call, exactly as FindOrIncrementError does for the same reopen.
|
|
319
|
+
# Without it this branch read occurrence_count into Ruby and wrote an
|
|
320
|
+
# ABSOLUTE value back, so two concurrent batches both read N and both
|
|
321
|
+
# wrote N+count -- one batch's events vanished while both reported
|
|
322
|
+
# success. The unresolved branch above is safe because it increments in
|
|
323
|
+
# SQL; this branch cannot use update_all because reopening is a state
|
|
324
|
+
# transition the dashboard must see, and update_all skips the
|
|
325
|
+
# after_update_commit broadcast.
|
|
326
|
+
resolved = resolved_scope.lock.order(last_seen_at: :desc).first
|
|
193
327
|
if resolved
|
|
328
|
+
# The count is incremented in SQL, never read into Ruby and written
|
|
329
|
+
# back. This branch used to compute `resolved.occurrence_count + count`
|
|
330
|
+
# and write that ABSOLUTE value, so two concurrent batches both read N
|
|
331
|
+
# and both wrote N+count -- one batch's events vanished while both
|
|
332
|
+
# reported success. The .lock above serializes the pair on
|
|
333
|
+
# PostgreSQL/MySQL; the atomic increment below conserves the count on
|
|
334
|
+
# every adapter, including SQLite where FOR UPDATE is a no-op.
|
|
335
|
+
ErrorLog.where(id: resolved.id).update_all([
|
|
336
|
+
"occurrence_count = occurrence_count + ?", count
|
|
337
|
+
])
|
|
338
|
+
|
|
339
|
+
# The reopen is a state transition the dashboard must see, so it stays
|
|
340
|
+
# an update! -- update_all would skip the after_update_commit
|
|
341
|
+
# broadcast. Reload first so this write does not clobber the increment
|
|
342
|
+
# just made with a stale in-memory occurrence_count.
|
|
343
|
+
resolved.reload
|
|
194
344
|
attrs = {
|
|
195
345
|
resolved: false,
|
|
196
346
|
status: "new",
|
|
197
347
|
resolved_at: nil,
|
|
198
|
-
occurrence_count: resolved.occurrence_count + count,
|
|
199
348
|
last_seen_at: last_seen
|
|
200
349
|
}
|
|
201
350
|
attrs[:reopened_at] = Time.current if ErrorLog.column_names.include?("reopened_at")
|
|
202
351
|
attrs[:environment] = env if env && resolved.environment.blank?
|
|
203
352
|
resolved.update!(attrs)
|
|
353
|
+
yield resolved.id if block_given?
|
|
204
354
|
return count
|
|
205
355
|
end
|
|
206
356
|
|
|
@@ -241,14 +391,67 @@ module RailsErrorDashboard
|
|
|
241
391
|
if ErrorLog.column_names.include?("context_captured_at")
|
|
242
392
|
create_attrs[:context_captured_at] = create_attrs[:occurred_at]
|
|
243
393
|
end
|
|
244
|
-
|
|
394
|
+
begin
|
|
395
|
+
# requires_new opens a SAVEPOINT: on PostgreSQL a failed INSERT aborts
|
|
396
|
+
# its transaction, and every later statement -- including the recovery
|
|
397
|
+
# lookup below -- fails with InFailedSqlTransaction. The savepoint
|
|
398
|
+
# confines the damage to this INSERT so the batch can continue.
|
|
399
|
+
ErrorLog.transaction(requires_new: true) do
|
|
400
|
+
created = ErrorLog.create!(**ErrorLog.clamp_string_attributes(Services::SensitiveDataFilter.filter_attributes(create_attrs)))
|
|
401
|
+
yield created.id if block_given?
|
|
402
|
+
end
|
|
403
|
+
rescue ActiveRecord::RecordNotUnique
|
|
404
|
+
# Another flush created this group between our lookups and this
|
|
405
|
+
# INSERT -- the group-identity index objected. Two concurrent batches
|
|
406
|
+
# for the same fingerprint is the NORMAL storm shape (every process
|
|
407
|
+
# flushes its own batch), so this must not fail the entry: the counts
|
|
408
|
+
# exist nowhere but in this payload. Re-run the same lookups and add
|
|
409
|
+
# to the row that now exists, exactly as FindOrIncrementError does.
|
|
410
|
+
#
|
|
411
|
+
# Nested in its own transaction because the failed INSERT poisons the
|
|
412
|
+
# surrounding one on PostgreSQL.
|
|
413
|
+
raise unless (target = existing_target(error_hash, application, env))
|
|
414
|
+
|
|
415
|
+
ErrorLog.where(id: target.id).update_all([
|
|
416
|
+
"occurrence_count = occurrence_count + ?, last_seen_at = ?", count, last_seen
|
|
417
|
+
])
|
|
418
|
+
# Yield on the RECOVERY path too: these counts are as real as the ones
|
|
419
|
+
# the winning INSERT wrote, so they need a time bucket as well, or a
|
|
420
|
+
# raced create silently loses its volume from every window query.
|
|
421
|
+
yield target.id if block_given?
|
|
422
|
+
end
|
|
245
423
|
count
|
|
246
424
|
end
|
|
247
425
|
|
|
426
|
+
# The row a retried INSERT should add to: any row holding this group
|
|
427
|
+
# identity, whatever its status. Deliberately wider than the
|
|
428
|
+
# priority-ordered lookups above -- the index has already proved a row
|
|
429
|
+
# with this identity exists, so refusing to match a resolved or wont_fix
|
|
430
|
+
# one would drop the counts instead.
|
|
431
|
+
def existing_target(error_hash, application, env)
|
|
432
|
+
scope = ErrorLog.where(error_hash: error_hash, application_id: application.id)
|
|
433
|
+
scope = scope.where(environment: [ env, nil ]) if env
|
|
434
|
+
scope.order(last_seen_at: :desc).select(:id).first
|
|
435
|
+
end
|
|
248
436
|
|
|
249
437
|
# The unresolved row the full capture path would increment right now.
|
|
438
|
+
# No time window: "won't fix" holds for as long as the row keeps the status.
|
|
439
|
+
def wont_fix_target(error_hash, application, env)
|
|
440
|
+
scope = ErrorLog
|
|
441
|
+
.where(error_hash: error_hash, application_id: application.id)
|
|
442
|
+
.where(status: "wont_fix")
|
|
443
|
+
if env
|
|
444
|
+
scope = scope.where(environment: [ env, nil ])
|
|
445
|
+
.order(Arel.sql("CASE WHEN environment IS NULL THEN 1 ELSE 0 END"))
|
|
446
|
+
end
|
|
447
|
+
scope.order(last_seen_at: :desc).select(:id, :environment).first
|
|
448
|
+
end
|
|
449
|
+
|
|
250
450
|
def unresolved_target(error_hash, application, env)
|
|
451
|
+
# Disjoint from wont_fix_target. Not where.not(...): that would also
|
|
452
|
+
# drop rows whose status is NULL.
|
|
251
453
|
scope = ErrorLog.unresolved
|
|
454
|
+
.where("status IS NULL OR status <> ?", "wont_fix")
|
|
252
455
|
.where(error_hash: error_hash, application_id: application.id)
|
|
253
456
|
.where("occurred_at >= ?", 24.hours.ago)
|
|
254
457
|
if env
|
|
@@ -303,6 +506,36 @@ module RailsErrorDashboard
|
|
|
303
506
|
Application.find_or_create_by_name(app_name)
|
|
304
507
|
end
|
|
305
508
|
|
|
509
|
+
# Record the interval whose per-event timing was lost.
|
|
510
|
+
#
|
|
511
|
+
# Keyed by the interval, NOT by a storm episode: the gate can shed events
|
|
512
|
+
# with its breaker closed and pass episode: nil, and a marker on the
|
|
513
|
+
# episode vanished in exactly that case. The events are just as
|
|
514
|
+
# untimed whether or not an episode object happens to exist.
|
|
515
|
+
#
|
|
516
|
+
# Bounds come from the entries themselves, so the gap describes when the
|
|
517
|
+
# events actually happened rather than when the worker got to them.
|
|
518
|
+
def record_timing_gap!(counted)
|
|
519
|
+
return unless EventTimingGap.table_exists?
|
|
520
|
+
|
|
521
|
+
times = @entries.filter_map do |entry|
|
|
522
|
+
entry = entry.with_indifferent_access if entry.respond_to?(:with_indifferent_access)
|
|
523
|
+
parse_time(entry["last_seen_at"]) || parse_time(entry["first_seen_at"])
|
|
524
|
+
end
|
|
525
|
+
first_seen = @entries.filter_map do |entry|
|
|
526
|
+
entry = entry.with_indifferent_access if entry.respond_to?(:with_indifferent_access)
|
|
527
|
+
parse_time(entry["first_seen_at"])
|
|
528
|
+
end
|
|
529
|
+
|
|
530
|
+
now = Time.current
|
|
531
|
+
EventTimingGap.create!(
|
|
532
|
+
application_id: resolve_application&.id,
|
|
533
|
+
covered_from: (first_seen + times).min || now,
|
|
534
|
+
covered_until: times.max || now,
|
|
535
|
+
events_affected: counted
|
|
536
|
+
)
|
|
537
|
+
end
|
|
538
|
+
|
|
306
539
|
def upsert_storm_event(counted)
|
|
307
540
|
return unless @episode.is_a?(Hash)
|
|
308
541
|
return unless StormEvent.table_exists?
|
|
@@ -323,6 +556,13 @@ module RailsErrorDashboard
|
|
|
323
556
|
event.fingerprints_affected = [ event.fingerprints_affected.to_i, @entries.size ].max
|
|
324
557
|
event.peak_rate_per_minute = [ event.peak_rate_per_minute.to_i, @episode["peak_rate_per_minute"].to_i ].max
|
|
325
558
|
event.reached_open ||= @episode["reached_open"] == true
|
|
559
|
+
# Sticky, like reached_open: once an episode has lost bucket timing it
|
|
560
|
+
# has lost it, and a later flush that happens to succeed does not make
|
|
561
|
+
# the earlier gap reappear. Guarded on the column so a host that has
|
|
562
|
+
# not run the migration yet keeps flushing normally.
|
|
563
|
+
if @buckets_incomplete && event.respond_to?(:buckets_incomplete)
|
|
564
|
+
event.buckets_incomplete = true
|
|
565
|
+
end
|
|
326
566
|
event.top_fingerprints = top_fingerprints_json(event)
|
|
327
567
|
event.ended_at = parse_time(@episode["ended_at"]) if @episode["ended_at"]
|
|
328
568
|
event.save!
|
|
@@ -25,8 +25,11 @@ module RailsErrorDashboard
|
|
|
25
25
|
app_id = current_application_id
|
|
26
26
|
|
|
27
27
|
# Process raise counts
|
|
28
|
+
# Keys are scrubbed before they are split: a class name or path with an
|
|
29
|
+
# invalid byte makes split/blank? raise, and the outer rescue would then
|
|
30
|
+
# drop every remaining count in the batch along with it.
|
|
28
31
|
@raise_counts.each do |key, count|
|
|
29
|
-
class_name, location = key.split("|", 2)
|
|
32
|
+
class_name, location = Services::EncodingSanitizer.scrub(key.to_s).split("|", 2)
|
|
30
33
|
next if class_name.blank? || location.blank?
|
|
31
34
|
|
|
32
35
|
upsert_raise(class_name, location, period, app_id, count)
|
|
@@ -34,7 +37,7 @@ module RailsErrorDashboard
|
|
|
34
37
|
|
|
35
38
|
# Process rescue counts
|
|
36
39
|
@rescue_counts.each do |key, count|
|
|
37
|
-
class_name, locations = key.split("|", 2)
|
|
40
|
+
class_name, locations = Services::EncodingSanitizer.scrub(key.to_s).split("|", 2)
|
|
38
41
|
next if class_name.blank? || locations.blank?
|
|
39
42
|
|
|
40
43
|
raise_loc, rescue_loc = locations.split("->", 2)
|
|
@@ -32,6 +32,7 @@ module RailsErrorDashboard
|
|
|
32
32
|
|
|
33
33
|
def call
|
|
34
34
|
return { success: false, error: red_t("red.commands.issue.url_required") } if @issue_url.blank?
|
|
35
|
+
return { success: false, error: red_t("red.commands.issue.url_invalid") } unless Services::UrlSafety.http_url?(@issue_url)
|
|
35
36
|
|
|
36
37
|
error = ErrorLog.find(@error_id)
|
|
37
38
|
parsed = parse_issue_url(@issue_url)
|