rails_error_dashboard 0.12.0 → 0.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/app/controllers/rails_error_dashboard/application_controller.rb +92 -11
- data/app/controllers/rails_error_dashboard/errors_controller.rb +111 -34
- data/app/controllers/rails_error_dashboard/webhooks_controller.rb +3 -0
- data/app/helpers/rails_error_dashboard/application_helper.rb +46 -1
- data/app/helpers/rails_error_dashboard/backtrace_helper.rb +8 -0
- data/app/jobs/rails_error_dashboard/concerns/plain_channel_message.rb +71 -0
- data/app/jobs/rails_error_dashboard/notification_burst_summary_job.rb +55 -0
- data/app/jobs/rails_error_dashboard/retention_cleanup_job.rb +68 -1
- data/app/jobs/rails_error_dashboard/storm_notification_job.rb +8 -44
- data/app/models/rails_error_dashboard/error_baseline.rb +12 -9
- data/app/models/rails_error_dashboard/error_comment.rb +0 -5
- data/app/models/rails_error_dashboard/error_log.rb +9 -3
- data/app/models/rails_error_dashboard/error_logs_record.rb +34 -0
- data/app/models/rails_error_dashboard/error_occurrence.rb +9 -1
- data/app/views/layouts/rails_error_dashboard.html.erb +11 -3
- data/app/views/rails_error_dashboard/errors/_discussion.html.erb +18 -11
- data/app/views/rails_error_dashboard/errors/_issue_section.html.erb +22 -8
- data/app/views/rails_error_dashboard/errors/_sidebar_metadata.html.erb +15 -6
- data/app/views/rails_error_dashboard/errors/analytics.html.erb +14 -8
- data/app/views/rails_error_dashboard/errors/diagnostic_dumps.html.erb +1 -1
- data/app/views/rails_error_dashboard/errors/index.html.erb +6 -1
- data/app/views/rails_error_dashboard/errors/platform_comparison.html.erb +7 -7
- data/app/views/rails_error_dashboard/errors/releases.html.erb +2 -2
- data/app/views/rails_error_dashboard/errors/settings.html.erb +2 -0
- data/config/locales/de.yml +28 -0
- data/config/locales/en.yml +35 -0
- data/config/locales/es.yml +28 -0
- data/config/locales/fr.yml +28 -0
- data/config/locales/it.yml +28 -0
- data/config/locales/ja.yml +28 -0
- data/config/locales/pl.yml +28 -0
- data/config/locales/pt-BR.yml +28 -0
- data/config/locales/ru.yml +28 -0
- data/config/locales/uk.yml +28 -0
- data/config/locales/zh-CN.yml +28 -0
- data/db/migrate/20260917000001_add_last_notified_at_to_error_logs.rb +40 -0
- data/lib/generators/rails_error_dashboard/install/templates/initializer.rb +12 -1
- data/lib/rails_error_dashboard/commands/assign_error.rb +12 -2
- data/lib/rails_error_dashboard/commands/backfill_environments.rb +2 -0
- data/lib/rails_error_dashboard/commands/backfill_resolved_at.rb +42 -0
- data/lib/rails_error_dashboard/commands/batch_delete_errors.rb +1 -0
- data/lib/rails_error_dashboard/commands/batch_mute_errors.rb +2 -0
- data/lib/rails_error_dashboard/commands/batch_resolve_errors.rb +2 -0
- data/lib/rails_error_dashboard/commands/batch_unmute_errors.rb +2 -0
- data/lib/rails_error_dashboard/commands/find_or_increment_error.rb +39 -7
- data/lib/rails_error_dashboard/commands/flush_rack_attack_events.rb +3 -1
- data/lib/rails_error_dashboard/commands/flush_storm_counts.rb +27 -4
- data/lib/rails_error_dashboard/commands/flush_swallowed_exceptions.rb +5 -2
- data/lib/rails_error_dashboard/commands/link_existing_issue.rb +1 -0
- data/lib/rails_error_dashboard/commands/log_error.rb +82 -23
- data/lib/rails_error_dashboard/commands/mute_error.rb +1 -0
- data/lib/rails_error_dashboard/commands/resolve_error.rb +2 -0
- data/lib/rails_error_dashboard/commands/scrub_invalid_encoding.rb +104 -0
- data/lib/rails_error_dashboard/commands/snooze_error.rb +34 -10
- data/lib/rails_error_dashboard/commands/unmute_error.rb +1 -0
- data/lib/rails_error_dashboard/commands/update_error_priority.rb +23 -2
- data/lib/rails_error_dashboard/commands/update_error_status.rb +33 -6
- data/lib/rails_error_dashboard/configuration.rb +19 -1
- data/lib/rails_error_dashboard/engine.rb +15 -0
- data/lib/rails_error_dashboard/queries/analytics_stats.rb +107 -26
- data/lib/rails_error_dashboard/queries/baseline_stats.rb +107 -0
- data/lib/rails_error_dashboard/queries/dashboard_stats.rb +55 -50
- data/lib/rails_error_dashboard/queries/error_correlation.rb +6 -3
- data/lib/rails_error_dashboard/queries/errors_list.rb +15 -2
- data/lib/rails_error_dashboard/queries/similar_errors.rb +1 -1
- data/lib/rails_error_dashboard/services/analytics_cache_manager.rb +43 -18
- data/lib/rails_error_dashboard/services/backtrace_processor.rb +3 -1
- data/lib/rails_error_dashboard/services/cause_chain_extractor.rb +3 -1
- data/lib/rails_error_dashboard/services/codeberg_issue_client.rb +13 -2
- data/lib/rails_error_dashboard/services/diagnostic_dump_generator.rb +5 -3
- data/lib/rails_error_dashboard/services/encoding_sanitizer.rb +80 -0
- data/lib/rails_error_dashboard/services/error_broadcaster.rb +180 -32
- data/lib/rails_error_dashboard/services/error_hash_generator.rb +9 -3
- data/lib/rails_error_dashboard/services/error_notification_dispatcher.rb +13 -0
- data/lib/rails_error_dashboard/services/exception_filter.rb +55 -0
- data/lib/rails_error_dashboard/services/git_head_reader.rb +102 -0
- data/lib/rails_error_dashboard/services/notification_throttler.rb +172 -28
- data/lib/rails_error_dashboard/services/sensitive_data_filter.rb +47 -1
- data/lib/rails_error_dashboard/services/storm_protection/circuit_breaker.rb +59 -3
- data/lib/rails_error_dashboard/services/storm_protection/fingerprint_buckets.rb +25 -2
- data/lib/rails_error_dashboard/services/storm_protection/gate.rb +32 -5
- data/lib/rails_error_dashboard/services/swallowed_exception_tracker.rb +99 -23
- data/lib/rails_error_dashboard/services/url_safety.rb +40 -0
- data/lib/rails_error_dashboard/subscribers/issue_tracker_subscriber.rb +11 -2
- data/lib/rails_error_dashboard/value_objects/error_context.rb +3 -1
- data/lib/rails_error_dashboard/version.rb +1 -1
- data/lib/rails_error_dashboard.rb +31 -0
- data/lib/tasks/error_dashboard.rake +54 -4
- metadata +10 -2
|
@@ -21,7 +21,10 @@ module RailsErrorDashboard
|
|
|
21
21
|
end
|
|
22
22
|
|
|
23
23
|
def initialize(entries:, overflow: 0, episode: nil, batch_id: nil)
|
|
24
|
-
|
|
24
|
+
# The gate scrubs what it buffers, so this is normally a no-op scan. It
|
|
25
|
+
# covers entries buffered by an older release and direct callers: the
|
|
26
|
+
# exemplar becomes an ErrorLog row, and its message is matched by regex.
|
|
27
|
+
@entries = Array(Services::EncodingSanitizer.scrub_deep(entries))
|
|
25
28
|
@overflow = overflow.to_i
|
|
26
29
|
@episode = episode
|
|
27
30
|
@batch_id = batch_id
|
|
@@ -165,7 +168,12 @@ module RailsErrorDashboard
|
|
|
165
168
|
# long-running unresolved error legitimately owns several unresolved
|
|
166
169
|
# rows. An update_all across the whole hash would add N to every one
|
|
167
170
|
# of them — seven real events becoming twelve counted occurrences.
|
|
168
|
-
|
|
171
|
+
#
|
|
172
|
+
# Priority 0, tried first: a wont_fix row absorbs its recurrences at any
|
|
173
|
+
# age and keeps its status (FindOrIncrementError#find_wont_fix). Same
|
|
174
|
+
# atomic increment, so it shares the branch below.
|
|
175
|
+
target = wont_fix_target(error_hash, application, env) ||
|
|
176
|
+
unresolved_target(error_hash, application, env)
|
|
169
177
|
if target
|
|
170
178
|
if env && target.environment.blank?
|
|
171
179
|
ErrorLog.where(id: target.id).update_all([
|
|
@@ -180,11 +188,11 @@ module RailsErrorDashboard
|
|
|
180
188
|
return count
|
|
181
189
|
end
|
|
182
190
|
|
|
183
|
-
# Priority 2: resolved
|
|
191
|
+
# Priority 2: resolved match — reopen, mirroring
|
|
184
192
|
# FindOrIncrementError so storm recurrences don't stay buried
|
|
185
193
|
resolved_scope = ErrorLog
|
|
186
194
|
.where(error_hash: error_hash, application_id: application.id)
|
|
187
|
-
.where(status:
|
|
195
|
+
.where(status: "resolved")
|
|
188
196
|
if env
|
|
189
197
|
resolved_scope = resolved_scope.where(environment: [ env, nil ])
|
|
190
198
|
.order(Arel.sql("CASE WHEN environment IS NULL THEN 1 ELSE 0 END"))
|
|
@@ -247,8 +255,23 @@ module RailsErrorDashboard
|
|
|
247
255
|
|
|
248
256
|
|
|
249
257
|
# The unresolved row the full capture path would increment right now.
|
|
258
|
+
# No time window: "won't fix" holds for as long as the row keeps the status.
|
|
259
|
+
def wont_fix_target(error_hash, application, env)
|
|
260
|
+
scope = ErrorLog
|
|
261
|
+
.where(error_hash: error_hash, application_id: application.id)
|
|
262
|
+
.where(status: "wont_fix")
|
|
263
|
+
if env
|
|
264
|
+
scope = scope.where(environment: [ env, nil ])
|
|
265
|
+
.order(Arel.sql("CASE WHEN environment IS NULL THEN 1 ELSE 0 END"))
|
|
266
|
+
end
|
|
267
|
+
scope.order(last_seen_at: :desc).select(:id, :environment).first
|
|
268
|
+
end
|
|
269
|
+
|
|
250
270
|
def unresolved_target(error_hash, application, env)
|
|
271
|
+
# Disjoint from wont_fix_target. Not where.not(...): that would also
|
|
272
|
+
# drop rows whose status is NULL.
|
|
251
273
|
scope = ErrorLog.unresolved
|
|
274
|
+
.where("status IS NULL OR status <> ?", "wont_fix")
|
|
252
275
|
.where(error_hash: error_hash, application_id: application.id)
|
|
253
276
|
.where("occurred_at >= ?", 24.hours.ago)
|
|
254
277
|
if env
|
|
@@ -25,8 +25,11 @@ module RailsErrorDashboard
|
|
|
25
25
|
app_id = current_application_id
|
|
26
26
|
|
|
27
27
|
# Process raise counts
|
|
28
|
+
# Keys are scrubbed before they are split: a class name or path with an
|
|
29
|
+
# invalid byte makes split/blank? raise, and the outer rescue would then
|
|
30
|
+
# drop every remaining count in the batch along with it.
|
|
28
31
|
@raise_counts.each do |key, count|
|
|
29
|
-
class_name, location = key.split("|", 2)
|
|
32
|
+
class_name, location = Services::EncodingSanitizer.scrub(key.to_s).split("|", 2)
|
|
30
33
|
next if class_name.blank? || location.blank?
|
|
31
34
|
|
|
32
35
|
upsert_raise(class_name, location, period, app_id, count)
|
|
@@ -34,7 +37,7 @@ module RailsErrorDashboard
|
|
|
34
37
|
|
|
35
38
|
# Process rescue counts
|
|
36
39
|
@rescue_counts.each do |key, count|
|
|
37
|
-
class_name, locations = key.split("|", 2)
|
|
40
|
+
class_name, locations = Services::EncodingSanitizer.scrub(key.to_s).split("|", 2)
|
|
38
41
|
next if class_name.blank? || locations.blank?
|
|
39
42
|
|
|
40
43
|
raise_loc, rescue_loc = locations.split("->", 2)
|
|
@@ -32,6 +32,7 @@ module RailsErrorDashboard
|
|
|
32
32
|
|
|
33
33
|
def call
|
|
34
34
|
return { success: false, error: red_t("red.commands.issue.url_required") } if @issue_url.blank?
|
|
35
|
+
return { success: false, error: red_t("red.commands.issue.url_invalid") } unless Services::UrlSafety.http_url?(@issue_url)
|
|
35
36
|
|
|
36
37
|
error = ErrorLog.find(@error_id)
|
|
37
38
|
parsed = parse_issue_url(@issue_url)
|
|
@@ -106,12 +106,18 @@ module RailsErrorDashboard
|
|
|
106
106
|
# FlushStormCounts#canonical_hash does for storm counts.
|
|
107
107
|
identity_parts = capture_identity_parts(exception, context)
|
|
108
108
|
|
|
109
|
-
|
|
109
|
+
# Scrub BEFORE anything is serialized: ActiveJob JSON-encodes the
|
|
110
|
+
# payload on this (the request) thread, and an invalid byte in the
|
|
111
|
+
# message, a backtrace line or the context would raise right here. The
|
|
112
|
+
# identity above is still taken from the raw exception, exactly as the
|
|
113
|
+
# sync path takes it, so grouping is unchanged.
|
|
114
|
+
context = Services::EncodingSanitizer.scrub_deep(context)
|
|
115
|
+
exception_data = Services::EncodingSanitizer.scrub_deep(
|
|
110
116
|
class_name: exception.class.name,
|
|
111
117
|
message: exception.message,
|
|
112
118
|
backtrace: exception.backtrace,
|
|
113
119
|
cause_chain: serialize_cause_chain(exception)
|
|
114
|
-
|
|
120
|
+
)
|
|
115
121
|
|
|
116
122
|
# Redact BEFORE the payload crosses the queue boundary. Until now the
|
|
117
123
|
# filter ran only just before the INSERT, so a durable adapter
|
|
@@ -264,6 +270,10 @@ module RailsErrorDashboard
|
|
|
264
270
|
|
|
265
271
|
context = context.merge(request_params: filtered[:request_params]) if context.key?(:request_params)
|
|
266
272
|
context = context.merge(request_url: filtered[:request_url]) if context.key?(:request_url)
|
|
273
|
+
# The raw session ID must not sit in Redis / Solid Queue either.
|
|
274
|
+
if context[:session_id]
|
|
275
|
+
context = context.merge(session_id: Services::SensitiveDataFilter.digest_session_id(context[:session_id]))
|
|
276
|
+
end
|
|
267
277
|
|
|
268
278
|
[ exception_data, context ]
|
|
269
279
|
rescue => e
|
|
@@ -292,7 +302,7 @@ module RailsErrorDashboard
|
|
|
292
302
|
chain << {
|
|
293
303
|
class_name: current.class.name,
|
|
294
304
|
message: current.message&.to_s,
|
|
295
|
-
backtrace: current.backtrace&.first(20)&.map { |line| Services::BacktraceProcessor.shorten_gem_path(line) }
|
|
305
|
+
backtrace: current.backtrace&.first(20)&.map { |line| Services::BacktraceProcessor.shorten_gem_path(Services::EncodingSanitizer.scrub(line)) }
|
|
296
306
|
}
|
|
297
307
|
|
|
298
308
|
current = current.respond_to?(:cause) ? current.cause : nil
|
|
@@ -317,7 +327,10 @@ module RailsErrorDashboard
|
|
|
317
327
|
# Job can retry it; every other failure is still swallowed.
|
|
318
328
|
def initialize(exception, context = {}, worker: false)
|
|
319
329
|
@exception = exception
|
|
320
|
-
|
|
330
|
+
# Invalid bytes anywhere in the context would raise as soon as it is
|
|
331
|
+
# JSON-encoded (ErrorContext does that in its constructor). A clean
|
|
332
|
+
# context costs one scan and its strings come back as the same objects.
|
|
333
|
+
@context = Services::EncodingSanitizer.scrub_deep(context)
|
|
321
334
|
@worker = worker
|
|
322
335
|
end
|
|
323
336
|
|
|
@@ -414,7 +427,7 @@ module RailsErrorDashboard
|
|
|
414
427
|
ENV["GIT_SHA"] ||
|
|
415
428
|
ENV["HEROKU_SLUG_COMMIT"] ||
|
|
416
429
|
ENV["RENDER_GIT_COMMIT"] ||
|
|
417
|
-
|
|
430
|
+
RailsErrorDashboard.detected_git_sha
|
|
418
431
|
end
|
|
419
432
|
|
|
420
433
|
if ErrorLog.column_names.include?("app_version")
|
|
@@ -434,6 +447,10 @@ module RailsErrorDashboard
|
|
|
434
447
|
attributes[:environment] = resolve_environment
|
|
435
448
|
end
|
|
436
449
|
|
|
450
|
+
# Neutralise invalid bytes BEFORE filtering: the filter runs regexes,
|
|
451
|
+
# which raise on an invalid string, and PostgreSQL rejects the INSERT.
|
|
452
|
+
attributes = Services::EncodingSanitizer.scrub_deep(attributes)
|
|
453
|
+
|
|
437
454
|
# Apply sensitive data filtering (on by default)
|
|
438
455
|
attributes = Services::SensitiveDataFilter.filter_attributes(attributes)
|
|
439
456
|
|
|
@@ -453,14 +470,14 @@ module RailsErrorDashboard
|
|
|
453
470
|
|
|
454
471
|
if raw_breadcrumbs.is_a?(Array) && raw_breadcrumbs.any?
|
|
455
472
|
filtered = Services::BreadcrumbCollector.filter_sensitive(raw_breadcrumbs)
|
|
456
|
-
attributes[:breadcrumbs] = filtered.to_json
|
|
473
|
+
attributes[:breadcrumbs] = Services::EncodingSanitizer.scrub_deep(filtered).to_json
|
|
457
474
|
end
|
|
458
475
|
end
|
|
459
476
|
|
|
460
477
|
# Capture system health snapshot (if enabled and column exists)
|
|
461
478
|
if !storm_lite && ErrorLog.column_names.include?("system_health") && RailsErrorDashboard.configuration.enable_system_health
|
|
462
479
|
health_data = @context[:_serialized_system_health] || Services::SystemHealthSnapshot.capture
|
|
463
|
-
attributes[:system_health] = health_data.to_json
|
|
480
|
+
attributes[:system_health] = Services::EncodingSanitizer.scrub_deep(health_data).to_json
|
|
464
481
|
end
|
|
465
482
|
|
|
466
483
|
# Capture local variables (if enabled and column exists)
|
|
@@ -472,7 +489,7 @@ module RailsErrorDashboard
|
|
|
472
489
|
raw_locals ||= @context[:_serialized_local_variables]
|
|
473
490
|
if raw_locals.is_a?(Hash) && raw_locals.any?
|
|
474
491
|
serialized = raw_locals == @context[:_serialized_local_variables] ? raw_locals : Services::VariableSerializer.call(raw_locals)
|
|
475
|
-
attributes[:local_variables] = serialized.to_json
|
|
492
|
+
attributes[:local_variables] = Services::EncodingSanitizer.scrub_deep(serialized).to_json
|
|
476
493
|
end
|
|
477
494
|
rescue => e
|
|
478
495
|
RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] Local variable serialization failed: #{e.message}")
|
|
@@ -496,7 +513,7 @@ module RailsErrorDashboard
|
|
|
496
513
|
additional_filter_patterns: RailsErrorDashboard.configuration.instance_variable_filter_patterns
|
|
497
514
|
)
|
|
498
515
|
end
|
|
499
|
-
attributes[:instance_variables] = serialized.to_json
|
|
516
|
+
attributes[:instance_variables] = Services::EncodingSanitizer.scrub_deep(serialized).to_json
|
|
500
517
|
end
|
|
501
518
|
rescue => e
|
|
502
519
|
RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] Instance variable serialization failed: #{e.message}")
|
|
@@ -532,13 +549,16 @@ module RailsErrorDashboard
|
|
|
532
549
|
occurred_at: attributes[:occurred_at],
|
|
533
550
|
user_id: attributes[:user_id],
|
|
534
551
|
request_id: error_context.request_id,
|
|
535
|
-
|
|
552
|
+
# Digest, not the raw ID -- it is a bearer credential. Idempotent,
|
|
553
|
+
# so a value already digested at the queue boundary passes through.
|
|
554
|
+
session_id: Services::SensitiveDataFilter.storable_session_id(error_context.session_id)
|
|
536
555
|
}
|
|
537
556
|
# The release THIS event happened under. The group keeps its first
|
|
538
557
|
# release; per-release counts come from here (ReleaseTimeline).
|
|
539
558
|
occurrence_columns = ErrorOccurrence.column_names
|
|
540
559
|
occurrence_attrs[:app_version] = attributes[:app_version] if occurrence_columns.include?("app_version")
|
|
541
560
|
occurrence_attrs[:git_sha] = attributes[:git_sha] if occurrence_columns.include?("git_sha")
|
|
561
|
+
occurrence_attrs = Services::EncodingSanitizer.scrub_deep(occurrence_attrs)
|
|
542
562
|
ErrorOccurrence.create(ErrorOccurrence.clamp_string_attributes(occurrence_attrs))
|
|
543
563
|
rescue => e
|
|
544
564
|
RailsErrorDashboard::Logger.error("Failed to create error occurrence: #{e.message}")
|
|
@@ -548,12 +568,12 @@ module RailsErrorDashboard
|
|
|
548
568
|
# Send notifications for new errors and reopened errors (with throttling).
|
|
549
569
|
# Muted errors skip notification dispatch but still fire plugin events.
|
|
550
570
|
if error_log.occurrence_count == 1
|
|
551
|
-
maybe_notify(error_log) { Services::NotificationThrottler.severity_meets_minimum?(error_log) }
|
|
571
|
+
maybe_notify(error_log, first_occurrence: true) { Services::NotificationThrottler.severity_meets_minimum?(error_log) }
|
|
552
572
|
PluginRegistry.dispatch(:on_error_logged, error_log)
|
|
553
573
|
trigger_callbacks(error_log)
|
|
554
574
|
emit_instrumentation_events(error_log)
|
|
555
575
|
elsif error_log.just_reopened
|
|
556
|
-
maybe_notify(error_log) { Services::NotificationThrottler.
|
|
576
|
+
maybe_notify(error_log, respect_cooldown: true) { Services::NotificationThrottler.severity_meets_minimum?(error_log) }
|
|
557
577
|
PluginRegistry.dispatch(:on_error_reopened, error_log)
|
|
558
578
|
trigger_callbacks(error_log)
|
|
559
579
|
emit_instrumentation_events(error_log)
|
|
@@ -590,14 +610,30 @@ module RailsErrorDashboard
|
|
|
590
610
|
# Muted errors skip notifications but still fire plugin events/callbacks.
|
|
591
611
|
# During a storm (breaker not closed) per-error notifications are
|
|
592
612
|
# suppressed — a single storm notification replaces them.
|
|
593
|
-
|
|
613
|
+
#
|
|
614
|
+
# respect_cooldown is true only for the reopened path, which is the only one
|
|
615
|
+
# the cooldown has ever applied to: a first occurrence and a threshold
|
|
616
|
+
# milestone always notify. They still stamp the row, so an error reopened
|
|
617
|
+
# minutes after its first notification is throttled.
|
|
618
|
+
def maybe_notify(error_log, respect_cooldown: false, first_occurrence: false)
|
|
594
619
|
return if error_log.muted?
|
|
620
|
+
# wont_fix: the team has decided not to act on this error, so its
|
|
621
|
+
# recurrences are counted and nothing else. Plugin events still fire,
|
|
622
|
+
# exactly as they do for a muted error.
|
|
623
|
+
return if error_log.status.to_s == "wont_fix"
|
|
595
624
|
return if Services::StormProtection::Gate.notifications_suppressed?
|
|
596
625
|
return unless Services::NotificationThrottler.environment_allowed?(error_log)
|
|
597
626
|
return unless yield
|
|
627
|
+
return if first_occurrence && burst_capped?
|
|
628
|
+
|
|
629
|
+
# Claim, THEN send. The claim is a conditional UPDATE only one process can
|
|
630
|
+
# win, so N workers reopening the same error send one notification, not
|
|
631
|
+
# N. The price: if the send below fails, this error is not retried inside
|
|
632
|
+
# the cooldown window. Recording after sending is what let every process
|
|
633
|
+
# through.
|
|
634
|
+
return unless Services::NotificationThrottler.claim!(error_log, respect_cooldown: respect_cooldown)
|
|
598
635
|
|
|
599
636
|
Services::ErrorNotificationDispatcher.call(error_log)
|
|
600
|
-
Services::NotificationThrottler.record_notification(error_log)
|
|
601
637
|
rescue => e
|
|
602
638
|
# The error row is already written by the time we get here. A channel
|
|
603
639
|
# that cannot be reached (Redis down for the Slack job's enqueue, a
|
|
@@ -608,6 +644,37 @@ module RailsErrorDashboard
|
|
|
608
644
|
)
|
|
609
645
|
end
|
|
610
646
|
|
|
647
|
+
# A bad deploy can produce hundreds of DISTINCT new errors, each a first
|
|
648
|
+
# occurrence the per-error cooldown never sees. Past
|
|
649
|
+
# config.notification_burst_limit per window, new-error notifications are
|
|
650
|
+
# held back and ONE summary says so. Asked only for first occurrences that
|
|
651
|
+
# were otherwise going to notify, and before the cooldown claim, so a
|
|
652
|
+
# suppressed error is not stamped as notified. The error itself is already
|
|
653
|
+
# stored; only the notification is dropped.
|
|
654
|
+
def burst_capped?
|
|
655
|
+
# Nothing can notify, so there is nothing to cap, and no summary to enqueue.
|
|
656
|
+
return false unless Services::ErrorNotificationDispatcher.any_channel?
|
|
657
|
+
|
|
658
|
+
case Services::NotificationThrottler.burst_decision
|
|
659
|
+
when :summarize
|
|
660
|
+
config = RailsErrorDashboard.configuration
|
|
661
|
+
NotificationBurstSummaryJob.perform_later(
|
|
662
|
+
limit: config.notification_burst_limit.to_i,
|
|
663
|
+
window_seconds: config.notification_burst_window_seconds.to_i,
|
|
664
|
+
locale: ApplicationJob.enqueue_locale
|
|
665
|
+
)
|
|
666
|
+
true
|
|
667
|
+
when :suppress
|
|
668
|
+
true
|
|
669
|
+
else
|
|
670
|
+
false
|
|
671
|
+
end
|
|
672
|
+
rescue => e
|
|
673
|
+
# Fail-open: a cap that cannot decide must not cost a notification.
|
|
674
|
+
RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] burst cap check failed: #{e.class}: #{e.message}")
|
|
675
|
+
false
|
|
676
|
+
end
|
|
677
|
+
|
|
611
678
|
# The environment this error is attributed to: an explicit context value
|
|
612
679
|
# (truncated to the column) or the process-wide resolution. Never nil.
|
|
613
680
|
def resolve_environment
|
|
@@ -704,6 +771,7 @@ module RailsErrorDashboard
|
|
|
704
771
|
# Return early if baseline alerts are disabled or error is muted
|
|
705
772
|
return unless config.enable_baseline_alerts
|
|
706
773
|
return if error_log.muted?
|
|
774
|
+
return if error_log.status.to_s == "wont_fix" # see maybe_notify
|
|
707
775
|
return unless Services::NotificationThrottler.environment_allowed?(error_log)
|
|
708
776
|
return unless defined?(Queries::BaselineStats)
|
|
709
777
|
return unless defined?(BaselineAlertJob)
|
|
@@ -760,15 +828,6 @@ module RailsErrorDashboard
|
|
|
760
828
|
nil
|
|
761
829
|
end
|
|
762
830
|
|
|
763
|
-
# Detect git SHA from git command (fallback)
|
|
764
|
-
def detect_git_sha_from_command
|
|
765
|
-
return nil unless File.exist?(Rails.root.join(".git"))
|
|
766
|
-
`git rev-parse --short HEAD 2>/dev/null`.strip.presence
|
|
767
|
-
rescue => e
|
|
768
|
-
RailsErrorDashboard::Logger.debug("Could not detect git SHA: #{e.message}")
|
|
769
|
-
nil
|
|
770
|
-
end
|
|
771
|
-
|
|
772
831
|
# Detect app version from VERSION file (fallback)
|
|
773
832
|
def detect_version_from_file
|
|
774
833
|
version_file = Rails.root.join("VERSION")
|
|
@@ -25,6 +25,8 @@ module RailsErrorDashboard
|
|
|
25
25
|
resolution_reference: @resolution_data[:resolution_reference],
|
|
26
26
|
status: "resolved"
|
|
27
27
|
)
|
|
28
|
+
# The stat cards are cached; a user action must show up at once.
|
|
29
|
+
Services::AnalyticsCacheManager.clear
|
|
28
30
|
|
|
29
31
|
# Dispatch plugin event for resolved error
|
|
30
32
|
PluginRegistry.dispatch(:on_error_resolved, error)
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RailsErrorDashboard
|
|
4
|
+
module Commands
|
|
5
|
+
# Command: repair rows that were stored with invalid bytes
|
|
6
|
+
#
|
|
7
|
+
# New captures are scrubbed on the way in (Services::EncodingSanitizer), but
|
|
8
|
+
# that does nothing for rows already in the table. SQLite and MySQL accept
|
|
9
|
+
# invalid UTF-8 and NUL bytes, and a row holding them makes the pages that
|
|
10
|
+
# render it answer 500. PostgreSQL rejects such rows at INSERT, so there is
|
|
11
|
+
# normally nothing to repair there; the scan is harmless.
|
|
12
|
+
#
|
|
13
|
+
# Behind `rake error_dashboard:scrub_invalid_encoding`. Safe to re-run: a
|
|
14
|
+
# second pass finds nothing to repair.
|
|
15
|
+
#
|
|
16
|
+
# @example
|
|
17
|
+
# ScrubInvalidEncoding.call # => { scanned: 1200, repaired: 3, unreadable: [] }
|
|
18
|
+
class ScrubInvalidEncoding
|
|
19
|
+
BATCH_SIZE = 500
|
|
20
|
+
|
|
21
|
+
MODEL_NAMES = %w[ErrorLog ErrorOccurrence].freeze
|
|
22
|
+
|
|
23
|
+
def self.call(batch_size: BATCH_SIZE)
|
|
24
|
+
new(batch_size: batch_size).call
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
def initialize(batch_size: BATCH_SIZE)
|
|
28
|
+
@batch_size = batch_size
|
|
29
|
+
@scanned = 0
|
|
30
|
+
@repaired = 0
|
|
31
|
+
@unreadable = []
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
# @return [Hash] { scanned:, repaired:, unreadable: [ "ErrorLog#12", ... ] }
|
|
35
|
+
def call
|
|
36
|
+
models.each { |model| scrub_model(model) }
|
|
37
|
+
|
|
38
|
+
{ scanned: @scanned, repaired: @repaired, unreadable: @unreadable }
|
|
39
|
+
end
|
|
40
|
+
|
|
41
|
+
private
|
|
42
|
+
|
|
43
|
+
def models
|
|
44
|
+
MODEL_NAMES.filter_map do |name|
|
|
45
|
+
model = RailsErrorDashboard.const_get(name)
|
|
46
|
+
model if model.table_exists?
|
|
47
|
+
rescue => e
|
|
48
|
+
RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] ScrubInvalidEncoding skipped #{name}: #{e.class}")
|
|
49
|
+
nil
|
|
50
|
+
end
|
|
51
|
+
end
|
|
52
|
+
|
|
53
|
+
def scrub_model(model)
|
|
54
|
+
columns = model.columns.select { |column| %i[string text].include?(column.type) }.map(&:name)
|
|
55
|
+
return if columns.empty?
|
|
56
|
+
|
|
57
|
+
model.in_batches(of: @batch_size) do |batch|
|
|
58
|
+
ids = batch.pluck(model.primary_key)
|
|
59
|
+
load_rows(model, ids).each { |row| scrub_row(row, columns) }
|
|
60
|
+
end
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
# One query per batch; if that batch cannot be materialised at all, fall
|
|
64
|
+
# back to row-by-row so a single unreadable row is reported by id instead
|
|
65
|
+
# of hiding the 499 readable ones around it.
|
|
66
|
+
def load_rows(model, ids)
|
|
67
|
+
model.where(model.primary_key => ids).to_a
|
|
68
|
+
rescue => e
|
|
69
|
+
RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] ScrubInvalidEncoding batch read failed: #{e.class}")
|
|
70
|
+
ids.filter_map do |id|
|
|
71
|
+
model.find(id)
|
|
72
|
+
rescue => row_error
|
|
73
|
+
@unreadable << "#{model.name.demodulize}##{id} (#{row_error.class})"
|
|
74
|
+
nil
|
|
75
|
+
end
|
|
76
|
+
end
|
|
77
|
+
|
|
78
|
+
def scrub_row(row, columns)
|
|
79
|
+
@scanned += 1
|
|
80
|
+
|
|
81
|
+
changes = columns.each_with_object({}) do |column, fixed|
|
|
82
|
+
value = row.read_attribute(column)
|
|
83
|
+
next unless value.is_a?(String)
|
|
84
|
+
|
|
85
|
+
clean = Services::EncodingSanitizer.scrub(value)
|
|
86
|
+
fixed[column] = clean unless clean.equal?(value) || clean == value
|
|
87
|
+
end
|
|
88
|
+
# The model scrubs invalid strings as it loads a row, so by now they
|
|
89
|
+
# read as clean. It remembers which ones it had to repair.
|
|
90
|
+
row.invalid_encoding_attributes.each do |column|
|
|
91
|
+
changes[column] = row.read_attribute(column) if columns.include?(column)
|
|
92
|
+
end
|
|
93
|
+
return if changes.empty?
|
|
94
|
+
|
|
95
|
+
# update_columns: no callbacks, no validations, no updated_at bump. This
|
|
96
|
+
# is a byte-level repair, not an edit.
|
|
97
|
+
row.update_columns(changes)
|
|
98
|
+
@repaired += 1
|
|
99
|
+
rescue => e
|
|
100
|
+
@unreadable << "#{row.class.name.demodulize}##{row.id} (#{e.class})"
|
|
101
|
+
end
|
|
102
|
+
end
|
|
103
|
+
end
|
|
104
|
+
end
|
|
@@ -4,7 +4,14 @@ module RailsErrorDashboard
|
|
|
4
4
|
module Commands
|
|
5
5
|
# Command: Snooze an error for a given number of hours
|
|
6
6
|
# This is a write operation that sets snoozed_until and optionally creates a comment
|
|
7
|
+
# Returns {success: bool, error: ErrorLog}; a failure also carries
|
|
8
|
+
# reason: :invalid_hours and writes nothing.
|
|
7
9
|
class SnoozeError
|
|
10
|
+
# 30 days. The form offers at most a week; the cap is for anything that
|
|
11
|
+
# does not come from the form. Without one, a negative value "snoozed"
|
|
12
|
+
# into the past and a huge one overflowed the timestamp.
|
|
13
|
+
MAX_SNOOZE_HOURS = 720
|
|
14
|
+
|
|
8
15
|
def self.call(error_id, hours:, reason: nil)
|
|
9
16
|
new(error_id, hours, reason).call
|
|
10
17
|
end
|
|
@@ -17,18 +24,35 @@ module RailsErrorDashboard
|
|
|
17
24
|
|
|
18
25
|
def call
|
|
19
26
|
error = ErrorLog.find(@error_id)
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
27
|
+
hours = whole_hours(@hours)
|
|
28
|
+
|
|
29
|
+
unless hours && (1..MAX_SNOOZE_HOURS).cover?(hours)
|
|
30
|
+
return { success: false, error: error, reason: :invalid_hours }
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
error.transaction do
|
|
34
|
+
if @reason.present?
|
|
35
|
+
error.comments.create!(
|
|
36
|
+
author_name: error.assigned_to || "System",
|
|
37
|
+
body: "Snoozed for #{hours} hours: #{@reason}"
|
|
38
|
+
)
|
|
39
|
+
end
|
|
40
|
+
|
|
41
|
+
error.update!(snoozed_until: hours.hours.from_now)
|
|
28
42
|
end
|
|
29
43
|
|
|
30
|
-
error
|
|
31
|
-
|
|
44
|
+
{ success: true, error: error }
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
private
|
|
48
|
+
|
|
49
|
+
# An Integer, or a String that is one ("24"). Anything else -- a Float, a
|
|
50
|
+
# nested parameter, "abc" -- is nil rather than a guess.
|
|
51
|
+
def whole_hours(value)
|
|
52
|
+
case value
|
|
53
|
+
when Integer then value
|
|
54
|
+
when String then Integer(value.strip, 10, exception: false)
|
|
55
|
+
end
|
|
32
56
|
end
|
|
33
57
|
end
|
|
34
58
|
end
|
|
@@ -4,6 +4,8 @@ module RailsErrorDashboard
|
|
|
4
4
|
module Commands
|
|
5
5
|
# Command: Update the priority level of an error
|
|
6
6
|
# This is a write operation that updates the priority_level field on an ErrorLog record
|
|
7
|
+
# Returns {success: bool, error: ErrorLog}; a failure also carries
|
|
8
|
+
# reason: :invalid_priority and leaves the existing priority alone.
|
|
7
9
|
class UpdateErrorPriority
|
|
8
10
|
def self.call(error_id, priority_level:)
|
|
9
11
|
new(error_id, priority_level).call
|
|
@@ -16,8 +18,27 @@ module RailsErrorDashboard
|
|
|
16
18
|
|
|
17
19
|
def call
|
|
18
20
|
error = ErrorLog.find(@error_id)
|
|
19
|
-
|
|
20
|
-
|
|
21
|
+
level = whole_number(@priority_level)
|
|
22
|
+
|
|
23
|
+
# The column is an integer, so an unchecked "x" was cast and stored as
|
|
24
|
+
# 0 -- silently replacing a real priority with Low.
|
|
25
|
+
unless ErrorLog::PRIORITY_LEVELS.key?(level)
|
|
26
|
+
return { success: false, error: error, reason: :invalid_priority }
|
|
27
|
+
end
|
|
28
|
+
|
|
29
|
+
error.update!(priority_level: level)
|
|
30
|
+
{ success: true, error: error }
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
private
|
|
34
|
+
|
|
35
|
+
# An Integer, or a String that is one ("3"). A nested parameter, a Float
|
|
36
|
+
# or free text is nil.
|
|
37
|
+
def whole_number(value)
|
|
38
|
+
case value
|
|
39
|
+
when Integer then value
|
|
40
|
+
when String then Integer(value.strip, 10, exception: false)
|
|
41
|
+
end
|
|
21
42
|
end
|
|
22
43
|
end
|
|
23
44
|
end
|
|
@@ -4,7 +4,8 @@ module RailsErrorDashboard
|
|
|
4
4
|
module Commands
|
|
5
5
|
# Command: Update the status of an error with optional comment
|
|
6
6
|
# This is a write operation that validates transitions and updates status
|
|
7
|
-
# Returns {success: bool, error: ErrorLog}
|
|
7
|
+
# Returns {success: bool, error: ErrorLog}; a failure also carries
|
|
8
|
+
# reason: :unknown_status or :invalid_transition so the caller can say which.
|
|
8
9
|
class UpdateErrorStatus
|
|
9
10
|
def self.call(error_id, status:, comment: nil)
|
|
10
11
|
new(error_id, status, comment).call
|
|
@@ -19,15 +20,17 @@ module RailsErrorDashboard
|
|
|
19
20
|
def call
|
|
20
21
|
error = ErrorLog.find(@error_id)
|
|
21
22
|
|
|
23
|
+
# A nested param arrives as a Hash-like object, never a known status.
|
|
24
|
+
unless @status.is_a?(String) && ErrorLog::STATUSES.include?(@status)
|
|
25
|
+
return { success: false, error: error, reason: :unknown_status }
|
|
26
|
+
end
|
|
27
|
+
|
|
22
28
|
unless error.can_transition_to?(@status)
|
|
23
|
-
return { success: false, error: error }
|
|
29
|
+
return { success: false, error: error, reason: :invalid_transition }
|
|
24
30
|
end
|
|
25
31
|
|
|
26
32
|
error.transaction do
|
|
27
|
-
error.update!(
|
|
28
|
-
|
|
29
|
-
# Auto-resolve if status is "resolved"
|
|
30
|
-
error.update!(resolved: true) if @status == "resolved"
|
|
33
|
+
error.update!(status_attributes(error))
|
|
31
34
|
|
|
32
35
|
# Add comment about status change
|
|
33
36
|
if @comment.present?
|
|
@@ -38,8 +41,32 @@ module RailsErrorDashboard
|
|
|
38
41
|
end
|
|
39
42
|
end
|
|
40
43
|
|
|
44
|
+
# The stat cards are cached; a user action must show up at once.
|
|
45
|
+
Services::AnalyticsCacheManager.clear
|
|
46
|
+
|
|
41
47
|
{ success: true, error: error }
|
|
42
48
|
end
|
|
49
|
+
|
|
50
|
+
private
|
|
51
|
+
|
|
52
|
+
# One write, so the three columns can never disagree. resolved_at is what
|
|
53
|
+
# MTTR is computed from: leaving it nil (as this command used to) dropped
|
|
54
|
+
# every error resolved through the status workflow from the MTTR figures,
|
|
55
|
+
# and leaving it set on a reopened error kept a stale resolution time.
|
|
56
|
+
# Only "resolved" sets the flag -- wont_fix stays resolved: false.
|
|
57
|
+
def status_attributes(error)
|
|
58
|
+
attrs = { status: @status }
|
|
59
|
+
|
|
60
|
+
if @status == "resolved"
|
|
61
|
+
attrs[:resolved] = true
|
|
62
|
+
attrs[:resolved_at] = Time.current
|
|
63
|
+
elsif error.status == "resolved" || error.resolved?
|
|
64
|
+
attrs[:resolved] = false
|
|
65
|
+
attrs[:resolved_at] = nil
|
|
66
|
+
end
|
|
67
|
+
|
|
68
|
+
attrs
|
|
69
|
+
end
|
|
43
70
|
end
|
|
44
71
|
end
|
|
45
72
|
end
|