rails_error_dashboard 0.12.1 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/app/controllers/rails_error_dashboard/application_controller.rb +92 -11
- data/app/controllers/rails_error_dashboard/errors_controller.rb +111 -34
- data/app/controllers/rails_error_dashboard/webhooks_controller.rb +3 -0
- data/app/helpers/rails_error_dashboard/application_helper.rb +46 -1
- data/app/helpers/rails_error_dashboard/backtrace_helper.rb +8 -0
- data/app/jobs/rails_error_dashboard/async_error_logging_job.rb +10 -0
- data/app/jobs/rails_error_dashboard/concerns/plain_channel_message.rb +71 -0
- data/app/jobs/rails_error_dashboard/notification_burst_summary_job.rb +55 -0
- data/app/jobs/rails_error_dashboard/retention_cleanup_job.rb +122 -1
- data/app/jobs/rails_error_dashboard/storm_flush_job.rb +7 -4
- data/app/jobs/rails_error_dashboard/storm_notification_job.rb +8 -44
- data/app/models/rails_error_dashboard/error_baseline.rb +12 -9
- data/app/models/rails_error_dashboard/error_comment.rb +0 -5
- data/app/models/rails_error_dashboard/error_log.rb +19 -3
- data/app/models/rails_error_dashboard/error_logs_record.rb +34 -0
- data/app/models/rails_error_dashboard/error_occurrence.rb +9 -1
- data/app/models/rails_error_dashboard/event_count.rb +132 -0
- data/app/models/rails_error_dashboard/event_timing_gap.rb +55 -0
- data/app/views/layouts/rails_error_dashboard.html.erb +58 -5
- data/app/views/rails_error_dashboard/errors/_discussion.html.erb +18 -11
- data/app/views/rails_error_dashboard/errors/_issue_section.html.erb +22 -8
- data/app/views/rails_error_dashboard/errors/_request_context.html.erb +2 -0
- data/app/views/rails_error_dashboard/errors/_sidebar_metadata.html.erb +15 -6
- data/app/views/rails_error_dashboard/errors/analytics.html.erb +8 -8
- data/app/views/rails_error_dashboard/errors/diagnostic_dumps.html.erb +1 -1
- data/app/views/rails_error_dashboard/errors/index.html.erb +6 -1
- data/app/views/rails_error_dashboard/errors/overview.html.erb +12 -0
- data/app/views/rails_error_dashboard/errors/platform_comparison.html.erb +7 -7
- data/app/views/rails_error_dashboard/errors/releases.html.erb +2 -2
- data/app/views/rails_error_dashboard/errors/settings.html.erb +2 -0
- data/app/views/rails_error_dashboard/errors/show.html.erb +3 -3
- data/config/locales/de.yml +30 -0
- data/config/locales/en.yml +37 -0
- data/config/locales/es.yml +30 -0
- data/config/locales/fr.yml +30 -0
- data/config/locales/it.yml +30 -0
- data/config/locales/ja.yml +30 -0
- data/config/locales/pl.yml +30 -0
- data/config/locales/pt-BR.yml +30 -0
- data/config/locales/ru.yml +30 -0
- data/config/locales/uk.yml +30 -0
- data/config/locales/zh-CN.yml +30 -0
- data/db/migrate/20260917000001_add_last_notified_at_to_error_logs.rb +40 -0
- data/db/migrate/20260919000001_create_event_counts.rb +71 -0
- data/db/migrate/20260920000001_add_buckets_incomplete_to_storm_events.rb +25 -0
- data/db/migrate/20260920000002_create_event_timing_gaps.rb +55 -0
- data/lib/generators/rails_error_dashboard/install/templates/initializer.rb +12 -1
- data/lib/rails_error_dashboard/commands/assign_error.rb +12 -2
- data/lib/rails_error_dashboard/commands/backfill_environments.rb +2 -0
- data/lib/rails_error_dashboard/commands/backfill_resolved_at.rb +42 -0
- data/lib/rails_error_dashboard/commands/batch_delete_errors.rb +1 -0
- data/lib/rails_error_dashboard/commands/batch_mute_errors.rb +2 -0
- data/lib/rails_error_dashboard/commands/batch_resolve_errors.rb +2 -0
- data/lib/rails_error_dashboard/commands/batch_unmute_errors.rb +2 -0
- data/lib/rails_error_dashboard/commands/find_or_increment_error.rb +109 -11
- data/lib/rails_error_dashboard/commands/flush_rack_attack_events.rb +3 -1
- data/lib/rails_error_dashboard/commands/flush_storm_counts.rb +252 -12
- data/lib/rails_error_dashboard/commands/flush_swallowed_exceptions.rb +5 -2
- data/lib/rails_error_dashboard/commands/link_existing_issue.rb +1 -0
- data/lib/rails_error_dashboard/commands/log_error.rb +291 -40
- data/lib/rails_error_dashboard/commands/mute_error.rb +1 -0
- data/lib/rails_error_dashboard/commands/resolve_error.rb +2 -0
- data/lib/rails_error_dashboard/commands/scrub_invalid_encoding.rb +104 -0
- data/lib/rails_error_dashboard/commands/snooze_error.rb +34 -10
- data/lib/rails_error_dashboard/commands/unmute_error.rb +1 -0
- data/lib/rails_error_dashboard/commands/update_error_priority.rb +23 -2
- data/lib/rails_error_dashboard/commands/update_error_status.rb +33 -6
- data/lib/rails_error_dashboard/configuration.rb +39 -1
- data/lib/rails_error_dashboard/engine.rb +28 -0
- data/lib/rails_error_dashboard/manual_error_reporter.rb +16 -5
- data/lib/rails_error_dashboard/queries/analytics_stats.rb +89 -30
- data/lib/rails_error_dashboard/queries/baseline_stats.rb +107 -0
- data/lib/rails_error_dashboard/queries/dashboard_stats.rb +216 -74
- data/lib/rails_error_dashboard/queries/error_correlation.rb +6 -3
- data/lib/rails_error_dashboard/queries/errors_list.rb +15 -2
- data/lib/rails_error_dashboard/queries/event_volume.rb +503 -0
- data/lib/rails_error_dashboard/queries/similar_errors.rb +1 -1
- data/lib/rails_error_dashboard/services/analytics_cache_manager.rb +43 -18
- data/lib/rails_error_dashboard/services/backtrace_processor.rb +3 -1
- data/lib/rails_error_dashboard/services/breadcrumb_collector.rb +23 -0
- data/lib/rails_error_dashboard/services/cause_chain_extractor.rb +3 -1
- data/lib/rails_error_dashboard/services/codeberg_issue_client.rb +13 -2
- data/lib/rails_error_dashboard/services/diagnostic_dump_generator.rb +5 -3
- data/lib/rails_error_dashboard/services/encoding_sanitizer.rb +80 -0
- data/lib/rails_error_dashboard/services/error_broadcaster.rb +180 -32
- data/lib/rails_error_dashboard/services/error_hash_generator.rb +9 -3
- data/lib/rails_error_dashboard/services/error_notification_dispatcher.rb +13 -0
- data/lib/rails_error_dashboard/services/exception_filter.rb +55 -0
- data/lib/rails_error_dashboard/services/git_head_reader.rb +102 -0
- data/lib/rails_error_dashboard/services/notification_throttler.rb +172 -28
- data/lib/rails_error_dashboard/services/sensitive_data_filter.rb +47 -1
- data/lib/rails_error_dashboard/services/storm_protection/circuit_breaker.rb +59 -3
- data/lib/rails_error_dashboard/services/storm_protection/count_buffer.rb +64 -6
- data/lib/rails_error_dashboard/services/storm_protection/fingerprint_buckets.rb +25 -2
- data/lib/rails_error_dashboard/services/storm_protection/gate.rb +32 -5
- data/lib/rails_error_dashboard/services/swallowed_exception_tracker.rb +99 -23
- data/lib/rails_error_dashboard/services/url_safety.rb +40 -0
- data/lib/rails_error_dashboard/services/variable_serializer.rb +125 -10
- data/lib/rails_error_dashboard/subscribers/breadcrumb_subscriber.rb +111 -0
- data/lib/rails_error_dashboard/subscribers/issue_tracker_subscriber.rb +11 -2
- data/lib/rails_error_dashboard/value_objects/error_context.rb +40 -3
- data/lib/rails_error_dashboard/version.rb +1 -1
- data/lib/rails_error_dashboard.rb +34 -0
- data/lib/tasks/error_dashboard.rake +54 -4
- metadata +16 -2
|
@@ -19,6 +19,14 @@ module RailsErrorDashboard
|
|
|
19
19
|
ActiveRecord::Deadlocked
|
|
20
20
|
].freeze
|
|
21
21
|
|
|
22
|
+
# Context keys whose Hash value ErrorContext#extract_params folds into the
|
|
23
|
+
# stored request_params. Each one must be redacted before the payload
|
|
24
|
+
# crosses the queue, or the queue and the database drift apart.
|
|
25
|
+
# :job/:job_class carry live objects rather than Hashes and are skipped by
|
|
26
|
+
# the is_a?(Hash) guard; their arguments reach the payload through
|
|
27
|
+
# :params, which is covered here.
|
|
28
|
+
CONTEXT_PARAM_KEYS = %i[params additional_context metadata].freeze
|
|
29
|
+
|
|
22
30
|
def self.call(exception, context = {})
|
|
23
31
|
# Filter FIRST (ignore list + static sampling) so ignored exceptions
|
|
24
32
|
# never count toward storm state. _pre_filtered prevents the sync path
|
|
@@ -79,7 +87,14 @@ module RailsErrorDashboard
|
|
|
79
87
|
# Kept as a module-level helper so both sync and async paths can call it.
|
|
80
88
|
# @return [Hash<String, Object>]
|
|
81
89
|
def self.build_capture_span_attributes(exception, was_async:)
|
|
82
|
-
|
|
90
|
+
# Redact BEFORE truncating. The span leaves the process for a collector
|
|
91
|
+
# the host app may not control, so it is an export boundary and gets the
|
|
92
|
+
# same policy as storage -- otherwise enabling tracing silently widened
|
|
93
|
+
# what counts as safe to emit. filter_attributes returns its input
|
|
94
|
+
# unchanged when filter_sensitive_data is off and rescues internally.
|
|
95
|
+
msg = Services::SensitiveDataFilter.filter_attributes(
|
|
96
|
+
message: exception.message.to_s
|
|
97
|
+
)[:message].to_s
|
|
83
98
|
{
|
|
84
99
|
"error.type" => exception.class.name,
|
|
85
100
|
"error.message" => msg.length > 200 ? "#{msg[0, 200]}…" : msg,
|
|
@@ -106,12 +121,18 @@ module RailsErrorDashboard
|
|
|
106
121
|
# FlushStormCounts#canonical_hash does for storm counts.
|
|
107
122
|
identity_parts = capture_identity_parts(exception, context)
|
|
108
123
|
|
|
109
|
-
|
|
124
|
+
# Scrub BEFORE anything is serialized: ActiveJob JSON-encodes the
|
|
125
|
+
# payload on this (the request) thread, and an invalid byte in the
|
|
126
|
+
# message, a backtrace line or the context would raise right here. The
|
|
127
|
+
# identity above is still taken from the raw exception, exactly as the
|
|
128
|
+
# sync path takes it, so grouping is unchanged.
|
|
129
|
+
context = Services::EncodingSanitizer.scrub_deep(context)
|
|
130
|
+
exception_data = Services::EncodingSanitizer.scrub_deep(
|
|
110
131
|
class_name: exception.class.name,
|
|
111
132
|
message: exception.message,
|
|
112
133
|
backtrace: exception.backtrace,
|
|
113
134
|
cause_chain: serialize_cause_chain(exception)
|
|
114
|
-
|
|
135
|
+
)
|
|
115
136
|
|
|
116
137
|
# Redact BEFORE the payload crosses the queue boundary. Until now the
|
|
117
138
|
# filter ran only just before the INSERT, so a durable adapter
|
|
@@ -127,13 +148,33 @@ module RailsErrorDashboard
|
|
|
127
148
|
exception_data, context = redact_async_payload(exception_data, context)
|
|
128
149
|
context = context.merge(_identity: identity_parts) if identity_parts
|
|
129
150
|
|
|
151
|
+
# Stamp WHEN and WHAT RELEASE this event was captured under, before it
|
|
152
|
+
# crosses the queue. Both used to be resolved by the worker from
|
|
153
|
+
# Time.current and its own process configuration, so a queue backed up
|
|
154
|
+
# across a deploy gave the event the drain time and the new release --
|
|
155
|
+
# an error captured at 12:00 under v1 and drained at 14:00 under v2 was
|
|
156
|
+
# stored as 14:00/v2, and release comparison blamed the wrong build.
|
|
157
|
+
# The row's created_at still records when the worker wrote it, so queue
|
|
158
|
+
# lag stays observable.
|
|
159
|
+
context = context.merge(_captured_at: (normalized_occurred_at(context) || Time.current).iso8601(6))
|
|
160
|
+
context = context.merge(_app_version: capture_app_version) unless context.key?(:_app_version)
|
|
161
|
+
context = context.merge(_git_sha: capture_git_sha) unless context.key?(:_git_sha)
|
|
162
|
+
|
|
130
163
|
# Storm shedding: :lite captures skip ALL pre-enqueue context harvest —
|
|
131
164
|
# this is request-thread CPU, the most valuable thing to shed.
|
|
132
165
|
lite = storm_lite?(context)
|
|
133
166
|
|
|
134
167
|
# Harvest breadcrumbs NOW (before job dispatch — different thread won't have them)
|
|
135
168
|
if !lite && RailsErrorDashboard.configuration.enable_breadcrumbs
|
|
136
|
-
|
|
169
|
+
# A failing job's snapshot rides the envelope: the worker that runs
|
|
170
|
+
# the capture is a different thread (and often a different process),
|
|
171
|
+
# so a thread-local left here would never be seen again. Prefer it
|
|
172
|
+
# over the live buffer, which the job's ensure has already cleared.
|
|
173
|
+
job_trail = Thread.current[Subscribers::BreadcrumbSubscriber::JOB_TRAIL_KEY]
|
|
174
|
+
Thread.current[Subscribers::BreadcrumbSubscriber::JOB_TRAIL_KEY] = nil
|
|
175
|
+
harvested = Services::BreadcrumbCollector.harvest
|
|
176
|
+
trail = job_trail.is_a?(Array) && job_trail.any? ? job_trail : harvested
|
|
177
|
+
context = context.merge(_serialized_breadcrumbs: trail)
|
|
137
178
|
end
|
|
138
179
|
|
|
139
180
|
# Capture system health NOW (metrics are time-sensitive, different thread = different state)
|
|
@@ -265,6 +306,28 @@ module RailsErrorDashboard
|
|
|
265
306
|
context = context.merge(request_params: filtered[:request_params]) if context.key?(:request_params)
|
|
266
307
|
context = context.merge(request_url: filtered[:request_url]) if context.key?(:request_url)
|
|
267
308
|
|
|
309
|
+
# request_params is only one of the shapes that becomes the stored
|
|
310
|
+
# params: ErrorContext#extract_params also folds in :params,
|
|
311
|
+
# :additional_context, :metadata and the job/sidekiq keys. Those crossed
|
|
312
|
+
# the queue raw, so the row was redacted while the secret sat in the
|
|
313
|
+
# queue's backing store -- exactly the drift this method exists to
|
|
314
|
+
# prevent. ParameterFilter takes a Hash directly; filter_json_string is
|
|
315
|
+
# no use here because these are Hashes, not JSON strings.
|
|
316
|
+
param_filter = Services::SensitiveDataFilter.parameter_filter
|
|
317
|
+
if param_filter
|
|
318
|
+
CONTEXT_PARAM_KEYS.each do |key|
|
|
319
|
+
value = context[key]
|
|
320
|
+
next unless value.is_a?(Hash)
|
|
321
|
+
|
|
322
|
+
context = context.merge(key => param_filter.filter(value))
|
|
323
|
+
end
|
|
324
|
+
end
|
|
325
|
+
|
|
326
|
+
# The raw session ID must not sit in Redis / Solid Queue either.
|
|
327
|
+
if context[:session_id]
|
|
328
|
+
context = context.merge(session_id: Services::SensitiveDataFilter.digest_session_id(context[:session_id]))
|
|
329
|
+
end
|
|
330
|
+
|
|
268
331
|
[ exception_data, context ]
|
|
269
332
|
rescue => e
|
|
270
333
|
# Never fail a capture over redaction. Fall back to the previous
|
|
@@ -292,7 +355,7 @@ module RailsErrorDashboard
|
|
|
292
355
|
chain << {
|
|
293
356
|
class_name: current.class.name,
|
|
294
357
|
message: current.message&.to_s,
|
|
295
|
-
backtrace: current.backtrace&.first(20)&.map { |line| Services::BacktraceProcessor.shorten_gem_path(line) }
|
|
358
|
+
backtrace: current.backtrace&.first(20)&.map { |line| Services::BacktraceProcessor.shorten_gem_path(Services::EncodingSanitizer.scrub(line)) }
|
|
296
359
|
}
|
|
297
360
|
|
|
298
361
|
current = current.respond_to?(:cause) ? current.cause : nil
|
|
@@ -317,7 +380,10 @@ module RailsErrorDashboard
|
|
|
317
380
|
# Job can retry it; every other failure is still swallowed.
|
|
318
381
|
def initialize(exception, context = {}, worker: false)
|
|
319
382
|
@exception = exception
|
|
320
|
-
|
|
383
|
+
# Invalid bytes anywhere in the context would raise as soon as it is
|
|
384
|
+
# JSON-encoded (ErrorContext does that in its constructor). A clean
|
|
385
|
+
# context costs one scan and its strings come back as the same objects.
|
|
386
|
+
@context = Services::EncodingSanitizer.scrub_deep(context)
|
|
321
387
|
@worker = worker
|
|
322
388
|
end
|
|
323
389
|
|
|
@@ -360,7 +426,11 @@ module RailsErrorDashboard
|
|
|
360
426
|
truncated_backtrace = Services::BacktraceProcessor.truncate(@exception.backtrace)
|
|
361
427
|
attributes = {
|
|
362
428
|
application_id: application.id,
|
|
363
|
-
|
|
429
|
+
# The reported type wins over the reconstructed class: an async
|
|
430
|
+
# capture of a type with no Ruby class here (a frontend error) is
|
|
431
|
+
# rebuilt as StandardError, and that name must not become the group
|
|
432
|
+
# identity. Falls back to the real class for every ordinary capture.
|
|
433
|
+
error_type: reported_error_type || @exception.class.name,
|
|
364
434
|
message: @exception.message,
|
|
365
435
|
backtrace: truncated_backtrace,
|
|
366
436
|
user_id: error_context.user_id,
|
|
@@ -371,7 +441,12 @@ module RailsErrorDashboard
|
|
|
371
441
|
platform: error_context.platform,
|
|
372
442
|
controller_name: error_context.controller_name,
|
|
373
443
|
action_name: error_context.action_name,
|
|
374
|
-
|
|
444
|
+
# Three sources, most specific first. A caller-supplied event time
|
|
445
|
+
# (ManualErrorReporter documents it, already clamped to not-future by
|
|
446
|
+
# ErrorContext) beats the capture-time stamp carried across the queue
|
|
447
|
+
# (see call_async), which in turn beats this worker's clock. Ordinary
|
|
448
|
+
# synchronous captures supply neither and fall through to now.
|
|
449
|
+
occurred_at: error_context.occurred_at || captured_at_from_context || Time.current
|
|
375
450
|
}
|
|
376
451
|
|
|
377
452
|
# Enriched request context (if columns exist)
|
|
@@ -410,17 +485,21 @@ module RailsErrorDashboard
|
|
|
410
485
|
|
|
411
486
|
# Add git/release info if columns exist
|
|
412
487
|
if ErrorLog.column_names.include?("git_sha")
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
ENV["RENDER_GIT_COMMIT"] ||
|
|
417
|
-
detect_git_sha_from_command
|
|
488
|
+
# The release the event was CAPTURED under, carried across the queue,
|
|
489
|
+
# falling back to this process's own for a synchronous capture.
|
|
490
|
+
attributes[:git_sha] = context_value(:_git_sha) || capture_git_sha
|
|
418
491
|
end
|
|
419
492
|
|
|
420
493
|
if ErrorLog.column_names.include?("app_version")
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
494
|
+
# Same precedence as occurred_at above. The reporter's own version
|
|
495
|
+
# wins over this server's -- for a mobile or frontend report they are
|
|
496
|
+
# different and the client's is the useful one (documented by
|
|
497
|
+
# ManualErrorReporter, previously discarded) -- then the release the
|
|
498
|
+
# event was CAPTURED under carried across the queue, then this
|
|
499
|
+
# process's own.
|
|
500
|
+
attributes[:app_version] = error_context.app_version ||
|
|
501
|
+
context_value(:_app_version) ||
|
|
502
|
+
capture_app_version
|
|
424
503
|
end
|
|
425
504
|
|
|
426
505
|
# Add environment snapshot (if column exists)
|
|
@@ -434,6 +513,10 @@ module RailsErrorDashboard
|
|
|
434
513
|
attributes[:environment] = resolve_environment
|
|
435
514
|
end
|
|
436
515
|
|
|
516
|
+
# Neutralise invalid bytes BEFORE filtering: the filter runs regexes,
|
|
517
|
+
# which raise on an invalid string, and PostgreSQL rejects the INSERT.
|
|
518
|
+
attributes = Services::EncodingSanitizer.scrub_deep(attributes)
|
|
519
|
+
|
|
437
520
|
# Apply sensitive data filtering (on by default)
|
|
438
521
|
attributes = Services::SensitiveDataFilter.filter_attributes(attributes)
|
|
439
522
|
|
|
@@ -442,25 +525,49 @@ module RailsErrorDashboard
|
|
|
442
525
|
|
|
443
526
|
# Harvest breadcrumbs (if enabled and column exists)
|
|
444
527
|
if !storm_lite && ErrorLog.column_names.include?("breadcrumbs") && RailsErrorDashboard.configuration.enable_breadcrumbs
|
|
445
|
-
#
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
#
|
|
449
|
-
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
|
|
528
|
+
# The envelope wins when there is one.
|
|
529
|
+
#
|
|
530
|
+
# An async capture harvests the REQUEST's trail before enqueue and
|
|
531
|
+
# carries it here; the worker thread running this job now has a
|
|
532
|
+
# buffer of its own (jobs get one, so a failing job has a trail), and
|
|
533
|
+
# harvesting that first would show the worker's activity in place of
|
|
534
|
+
# the request's. Draining the current thread stays the sync path, and
|
|
535
|
+
# still runs below so a worker's own buffer is not left to leak.
|
|
536
|
+
serialized = @context[:_serialized_breadcrumbs] || @context["_serialized_breadcrumbs"]
|
|
537
|
+
|
|
538
|
+
# A failing job's trail, snapshotted by the around_perform on its way
|
|
539
|
+
# out. Active Job reports a job error two frames OUTSIDE the callback
|
|
540
|
+
# that owns the buffer, so by now the live buffer is already gone and
|
|
541
|
+
# a current-thread harvest returns nothing -- the snapshot is the only
|
|
542
|
+
# surviving copy. Consumed here, whoever wrote it.
|
|
543
|
+
job_trail = Thread.current[Subscribers::BreadcrumbSubscriber::JOB_TRAIL_KEY]
|
|
544
|
+
Thread.current[Subscribers::BreadcrumbSubscriber::JOB_TRAIL_KEY] = nil
|
|
545
|
+
|
|
546
|
+
# Still unconditional: drains a worker's own buffer so it cannot leak.
|
|
547
|
+
current = Services::BreadcrumbCollector.harvest
|
|
548
|
+
|
|
549
|
+
# The envelope stays FIRST. An async capture carries the request's
|
|
550
|
+
# trail across the queue, and neither the worker's own buffer nor a
|
|
551
|
+
# snapshot may displace it.
|
|
552
|
+
raw_breadcrumbs =
|
|
553
|
+
if serialized.is_a?(Array) && serialized.any?
|
|
554
|
+
serialized
|
|
555
|
+
elsif job_trail.is_a?(Array) && job_trail.any?
|
|
556
|
+
job_trail
|
|
557
|
+
else
|
|
558
|
+
current
|
|
559
|
+
end
|
|
453
560
|
|
|
454
561
|
if raw_breadcrumbs.is_a?(Array) && raw_breadcrumbs.any?
|
|
455
562
|
filtered = Services::BreadcrumbCollector.filter_sensitive(raw_breadcrumbs)
|
|
456
|
-
attributes[:breadcrumbs] = filtered.to_json
|
|
563
|
+
attributes[:breadcrumbs] = Services::EncodingSanitizer.scrub_deep(filtered).to_json
|
|
457
564
|
end
|
|
458
565
|
end
|
|
459
566
|
|
|
460
567
|
# Capture system health snapshot (if enabled and column exists)
|
|
461
568
|
if !storm_lite && ErrorLog.column_names.include?("system_health") && RailsErrorDashboard.configuration.enable_system_health
|
|
462
569
|
health_data = @context[:_serialized_system_health] || Services::SystemHealthSnapshot.capture
|
|
463
|
-
attributes[:system_health] = health_data.to_json
|
|
570
|
+
attributes[:system_health] = Services::EncodingSanitizer.scrub_deep(health_data).to_json
|
|
464
571
|
end
|
|
465
572
|
|
|
466
573
|
# Capture local variables (if enabled and column exists)
|
|
@@ -472,7 +579,7 @@ module RailsErrorDashboard
|
|
|
472
579
|
raw_locals ||= @context[:_serialized_local_variables]
|
|
473
580
|
if raw_locals.is_a?(Hash) && raw_locals.any?
|
|
474
581
|
serialized = raw_locals == @context[:_serialized_local_variables] ? raw_locals : Services::VariableSerializer.call(raw_locals)
|
|
475
|
-
attributes[:local_variables] = serialized.to_json
|
|
582
|
+
attributes[:local_variables] = Services::EncodingSanitizer.scrub_deep(serialized).to_json
|
|
476
583
|
end
|
|
477
584
|
rescue => e
|
|
478
585
|
RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] Local variable serialization failed: #{e.message}")
|
|
@@ -496,7 +603,7 @@ module RailsErrorDashboard
|
|
|
496
603
|
additional_filter_patterns: RailsErrorDashboard.configuration.instance_variable_filter_patterns
|
|
497
604
|
)
|
|
498
605
|
end
|
|
499
|
-
attributes[:instance_variables] = serialized.to_json
|
|
606
|
+
attributes[:instance_variables] = Services::EncodingSanitizer.scrub_deep(serialized).to_json
|
|
500
607
|
end
|
|
501
608
|
rescue => e
|
|
502
609
|
RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] Instance variable serialization failed: #{e.message}")
|
|
@@ -511,9 +618,15 @@ module RailsErrorDashboard
|
|
|
511
618
|
# context payloads by design, and must not be recorded as though it
|
|
512
619
|
# refreshed the snapshot -- nor allowed to overwrite a good backtrace.
|
|
513
620
|
# It is stripped before the INSERT (it is a signal, not a column).
|
|
621
|
+
#
|
|
622
|
+
# Only the SHED case is asserted here. "This capture was complete" and
|
|
623
|
+
# "the stored snapshot is complete" are different claims: when the
|
|
624
|
+
# grouping command keeps an earlier occurrence's user or locals beside
|
|
625
|
+
# this one's URL, the row displays a mixture, and only that command can
|
|
626
|
+
# see it. Passing nil lets it decide between "full" and "partial".
|
|
514
627
|
error_log = ErrorLog.find_or_increment_by_hash(
|
|
515
628
|
error_hash,
|
|
516
|
-
attributes.merge(error_hash: error_hash, _context_fidelity: storm_lite ? "lite" :
|
|
629
|
+
attributes.merge(error_hash: error_hash, _context_fidelity: storm_lite ? "lite" : nil)
|
|
517
630
|
)
|
|
518
631
|
|
|
519
632
|
# OTel: now that the error_log exists, attach its id + dedup flag + severity
|
|
@@ -532,13 +645,16 @@ module RailsErrorDashboard
|
|
|
532
645
|
occurred_at: attributes[:occurred_at],
|
|
533
646
|
user_id: attributes[:user_id],
|
|
534
647
|
request_id: error_context.request_id,
|
|
535
|
-
|
|
648
|
+
# Digest, not the raw ID -- it is a bearer credential. Idempotent,
|
|
649
|
+
# so a value already digested at the queue boundary passes through.
|
|
650
|
+
session_id: Services::SensitiveDataFilter.storable_session_id(error_context.session_id)
|
|
536
651
|
}
|
|
537
652
|
# The release THIS event happened under. The group keeps its first
|
|
538
653
|
# release; per-release counts come from here (ReleaseTimeline).
|
|
539
654
|
occurrence_columns = ErrorOccurrence.column_names
|
|
540
655
|
occurrence_attrs[:app_version] = attributes[:app_version] if occurrence_columns.include?("app_version")
|
|
541
656
|
occurrence_attrs[:git_sha] = attributes[:git_sha] if occurrence_columns.include?("git_sha")
|
|
657
|
+
occurrence_attrs = Services::EncodingSanitizer.scrub_deep(occurrence_attrs)
|
|
542
658
|
ErrorOccurrence.create(ErrorOccurrence.clamp_string_attributes(occurrence_attrs))
|
|
543
659
|
rescue => e
|
|
544
660
|
RailsErrorDashboard::Logger.error("Failed to create error occurrence: #{e.message}")
|
|
@@ -548,12 +664,12 @@ module RailsErrorDashboard
|
|
|
548
664
|
# Send notifications for new errors and reopened errors (with throttling).
|
|
549
665
|
# Muted errors skip notification dispatch but still fire plugin events.
|
|
550
666
|
if error_log.occurrence_count == 1
|
|
551
|
-
maybe_notify(error_log) { Services::NotificationThrottler.severity_meets_minimum?(error_log) }
|
|
667
|
+
maybe_notify(error_log, first_occurrence: true) { Services::NotificationThrottler.severity_meets_minimum?(error_log) }
|
|
552
668
|
PluginRegistry.dispatch(:on_error_logged, error_log)
|
|
553
669
|
trigger_callbacks(error_log)
|
|
554
670
|
emit_instrumentation_events(error_log)
|
|
555
671
|
elsif error_log.just_reopened
|
|
556
|
-
maybe_notify(error_log) { Services::NotificationThrottler.
|
|
672
|
+
maybe_notify(error_log, respect_cooldown: true) { Services::NotificationThrottler.severity_meets_minimum?(error_log) }
|
|
557
673
|
PluginRegistry.dispatch(:on_error_reopened, error_log)
|
|
558
674
|
trigger_callbacks(error_log)
|
|
559
675
|
emit_instrumentation_events(error_log)
|
|
@@ -590,14 +706,30 @@ module RailsErrorDashboard
|
|
|
590
706
|
# Muted errors skip notifications but still fire plugin events/callbacks.
|
|
591
707
|
# During a storm (breaker not closed) per-error notifications are
|
|
592
708
|
# suppressed — a single storm notification replaces them.
|
|
593
|
-
|
|
709
|
+
#
|
|
710
|
+
# respect_cooldown is true only for the reopened path, which is the only one
|
|
711
|
+
# the cooldown has ever applied to: a first occurrence and a threshold
|
|
712
|
+
# milestone always notify. They still stamp the row, so an error reopened
|
|
713
|
+
# minutes after its first notification is throttled.
|
|
714
|
+
def maybe_notify(error_log, respect_cooldown: false, first_occurrence: false)
|
|
594
715
|
return if error_log.muted?
|
|
716
|
+
# wont_fix: the team has decided not to act on this error, so its
|
|
717
|
+
# recurrences are counted and nothing else. Plugin events still fire,
|
|
718
|
+
# exactly as they do for a muted error.
|
|
719
|
+
return if error_log.status.to_s == "wont_fix"
|
|
595
720
|
return if Services::StormProtection::Gate.notifications_suppressed?
|
|
596
721
|
return unless Services::NotificationThrottler.environment_allowed?(error_log)
|
|
597
722
|
return unless yield
|
|
723
|
+
return if first_occurrence && burst_capped?
|
|
724
|
+
|
|
725
|
+
# Claim, THEN send. The claim is a conditional UPDATE only one process can
|
|
726
|
+
# win, so N workers reopening the same error send one notification, not
|
|
727
|
+
# N. The price: if the send below fails, this error is not retried inside
|
|
728
|
+
# the cooldown window. Recording after sending is what let every process
|
|
729
|
+
# through.
|
|
730
|
+
return unless Services::NotificationThrottler.claim!(error_log, respect_cooldown: respect_cooldown)
|
|
598
731
|
|
|
599
732
|
Services::ErrorNotificationDispatcher.call(error_log)
|
|
600
|
-
Services::NotificationThrottler.record_notification(error_log)
|
|
601
733
|
rescue => e
|
|
602
734
|
# The error row is already written by the time we get here. A channel
|
|
603
735
|
# that cannot be reached (Redis down for the Slack job's enqueue, a
|
|
@@ -608,6 +740,37 @@ module RailsErrorDashboard
|
|
|
608
740
|
)
|
|
609
741
|
end
|
|
610
742
|
|
|
743
|
+
# A bad deploy can produce hundreds of DISTINCT new errors, each a first
|
|
744
|
+
# occurrence the per-error cooldown never sees. Past
|
|
745
|
+
# config.notification_burst_limit per window, new-error notifications are
|
|
746
|
+
# held back and ONE summary says so. Asked only for first occurrences that
|
|
747
|
+
# were otherwise going to notify, and before the cooldown claim, so a
|
|
748
|
+
# suppressed error is not stamped as notified. The error itself is already
|
|
749
|
+
# stored; only the notification is dropped.
|
|
750
|
+
def burst_capped?
|
|
751
|
+
# Nothing can notify, so there is nothing to cap, and no summary to enqueue.
|
|
752
|
+
return false unless Services::ErrorNotificationDispatcher.any_channel?
|
|
753
|
+
|
|
754
|
+
case Services::NotificationThrottler.burst_decision
|
|
755
|
+
when :summarize
|
|
756
|
+
config = RailsErrorDashboard.configuration
|
|
757
|
+
NotificationBurstSummaryJob.perform_later(
|
|
758
|
+
limit: config.notification_burst_limit.to_i,
|
|
759
|
+
window_seconds: config.notification_burst_window_seconds.to_i,
|
|
760
|
+
locale: ApplicationJob.enqueue_locale
|
|
761
|
+
)
|
|
762
|
+
true
|
|
763
|
+
when :suppress
|
|
764
|
+
true
|
|
765
|
+
else
|
|
766
|
+
false
|
|
767
|
+
end
|
|
768
|
+
rescue => e
|
|
769
|
+
# Fail-open: a cap that cannot decide must not cost a notification.
|
|
770
|
+
RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] burst cap check failed: #{e.class}: #{e.message}")
|
|
771
|
+
false
|
|
772
|
+
end
|
|
773
|
+
|
|
611
774
|
# The environment this error is attributed to: an explicit context value
|
|
612
775
|
# (truncated to the column) or the process-wide resolution. Never nil.
|
|
613
776
|
def resolve_environment
|
|
@@ -704,6 +867,7 @@ module RailsErrorDashboard
|
|
|
704
867
|
# Return early if baseline alerts are disabled or error is muted
|
|
705
868
|
return unless config.enable_baseline_alerts
|
|
706
869
|
return if error_log.muted?
|
|
870
|
+
return if error_log.status.to_s == "wont_fix" # see maybe_notify
|
|
707
871
|
return unless Services::NotificationThrottler.environment_allowed?(error_log)
|
|
708
872
|
return unless defined?(Queries::BaselineStats)
|
|
709
873
|
return unless defined?(BaselineAlertJob)
|
|
@@ -760,17 +924,75 @@ module RailsErrorDashboard
|
|
|
760
924
|
nil
|
|
761
925
|
end
|
|
762
926
|
|
|
763
|
-
#
|
|
764
|
-
|
|
765
|
-
|
|
766
|
-
|
|
767
|
-
|
|
768
|
-
|
|
927
|
+
# The error type as REPORTED, carried across the queue by
|
|
928
|
+
# AsyncErrorLoggingJob. Symbol or string key: ActiveJob's serializer
|
|
929
|
+
# turns symbol keys into strings on the way through.
|
|
930
|
+
def reported_error_type
|
|
931
|
+
return nil unless @context.is_a?(Hash)
|
|
932
|
+
|
|
933
|
+
(@context[:_reported_error_type] || @context["_reported_error_type"]).presence
|
|
934
|
+
rescue StandardError
|
|
769
935
|
nil
|
|
770
936
|
end
|
|
771
937
|
|
|
772
938
|
# Detect app version from VERSION file (fallback)
|
|
773
939
|
def detect_version_from_file
|
|
940
|
+
self.class.detect_version_from_file
|
|
941
|
+
end
|
|
942
|
+
|
|
943
|
+
# --- Capture-time envelope -------------------------------------------
|
|
944
|
+
#
|
|
945
|
+
# The release THIS process is running, resolved on the capture thread so
|
|
946
|
+
# it can be stamped onto the payload before it crosses the queue. The
|
|
947
|
+
# worker reads the stamp instead of asking its own configuration, which
|
|
948
|
+
# is what made a queued event inherit the release it was drained under.
|
|
949
|
+
|
|
950
|
+
# ONE normalization of a caller-supplied event time, shared by both
|
|
951
|
+
# transports.
|
|
952
|
+
#
|
|
953
|
+
# ErrorContext#extract_occurred_at already parses Strings, clamps a
|
|
954
|
+
# future time and rescues bad input -- but it runs in the WORKER on the
|
|
955
|
+
# async path, long after this method's caller has already had to
|
|
956
|
+
# serialize the value. Stamping the envelope by calling .iso8601 on the
|
|
957
|
+
# raw input therefore raised NoMethodError for a String (a documented
|
|
958
|
+
# ManualErrorReporter input form), the outer rescue in .call swallowed
|
|
959
|
+
# it, and the capture vanished: nothing enqueued, no row, nothing logged
|
|
960
|
+
# at error level. The sync path accepted the identical input.
|
|
961
|
+
#
|
|
962
|
+
# Normalizing here keeps ONE policy -- including the future clamp -- and
|
|
963
|
+
# returns nil rather than raising, so an unparseable value costs the
|
|
964
|
+
# timestamp and never the error itself (Safety Rule 1).
|
|
965
|
+
# @return [Time, nil]
|
|
966
|
+
def self.normalized_occurred_at(context)
|
|
967
|
+
raw = context[:occurred_at] || context["occurred_at"]
|
|
968
|
+
return nil if raw.nil? || (raw.respond_to?(:empty?) && raw.empty?)
|
|
969
|
+
|
|
970
|
+
time = raw.is_a?(String) ? Time.zone.parse(raw) : raw
|
|
971
|
+
return nil unless time.respond_to?(:to_time)
|
|
972
|
+
|
|
973
|
+
[ time.to_time, Time.current ].min
|
|
974
|
+
rescue StandardError => e
|
|
975
|
+
RailsErrorDashboard::Logger.debug(
|
|
976
|
+
"[RailsErrorDashboard] Unparseable occurred_at (#{e.class}); using capture time"
|
|
977
|
+
)
|
|
978
|
+
nil
|
|
979
|
+
end
|
|
980
|
+
|
|
981
|
+
def self.capture_app_version
|
|
982
|
+
RailsErrorDashboard.configuration.app_version ||
|
|
983
|
+
ENV["APP_VERSION"] ||
|
|
984
|
+
detect_version_from_file
|
|
985
|
+
end
|
|
986
|
+
|
|
987
|
+
def self.capture_git_sha
|
|
988
|
+
RailsErrorDashboard.configuration.git_sha ||
|
|
989
|
+
ENV["GIT_SHA"] ||
|
|
990
|
+
ENV["HEROKU_SLUG_COMMIT"] ||
|
|
991
|
+
ENV["RENDER_GIT_COMMIT"] ||
|
|
992
|
+
RailsErrorDashboard.detected_git_sha
|
|
993
|
+
end
|
|
994
|
+
|
|
995
|
+
def self.detect_version_from_file
|
|
774
996
|
version_file = Rails.root.join("VERSION")
|
|
775
997
|
return File.read(version_file).strip if File.exist?(version_file)
|
|
776
998
|
nil
|
|
@@ -778,6 +1000,35 @@ module RailsErrorDashboard
|
|
|
778
1000
|
RailsErrorDashboard::Logger.debug("Could not detect version: #{e.message}")
|
|
779
1001
|
nil
|
|
780
1002
|
end
|
|
1003
|
+
|
|
1004
|
+
# Symbol or string key: the async job round-trips the context through the
|
|
1005
|
+
# queue serializer, which turns symbol keys into strings.
|
|
1006
|
+
def context_value(key)
|
|
1007
|
+
return nil unless @context.is_a?(Hash)
|
|
1008
|
+
|
|
1009
|
+
(@context[key] || @context[key.to_s]).presence
|
|
1010
|
+
rescue StandardError
|
|
1011
|
+
nil
|
|
1012
|
+
end
|
|
1013
|
+
|
|
1014
|
+
def capture_app_version
|
|
1015
|
+
self.class.capture_app_version
|
|
1016
|
+
end
|
|
1017
|
+
|
|
1018
|
+
def capture_git_sha
|
|
1019
|
+
self.class.capture_git_sha
|
|
1020
|
+
end
|
|
1021
|
+
|
|
1022
|
+
# The capture-time stamp an async payload carries, or nil for a
|
|
1023
|
+
# synchronous capture (which has no queue hop and is already "now").
|
|
1024
|
+
def captured_at_from_context
|
|
1025
|
+
raw = context_value(:_captured_at)
|
|
1026
|
+
return nil if raw.blank?
|
|
1027
|
+
|
|
1028
|
+
Time.zone ? Time.zone.parse(raw.to_s) : Time.parse(raw.to_s)
|
|
1029
|
+
rescue StandardError
|
|
1030
|
+
nil
|
|
1031
|
+
end
|
|
781
1032
|
end
|
|
782
1033
|
end
|
|
783
1034
|
end
|
|
@@ -25,6 +25,8 @@ module RailsErrorDashboard
|
|
|
25
25
|
resolution_reference: @resolution_data[:resolution_reference],
|
|
26
26
|
status: "resolved"
|
|
27
27
|
)
|
|
28
|
+
# The stat cards are cached; a user action must show up at once.
|
|
29
|
+
Services::AnalyticsCacheManager.clear
|
|
28
30
|
|
|
29
31
|
# Dispatch plugin event for resolved error
|
|
30
32
|
PluginRegistry.dispatch(:on_error_resolved, error)
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RailsErrorDashboard
|
|
4
|
+
module Commands
|
|
5
|
+
# Command: repair rows that were stored with invalid bytes
|
|
6
|
+
#
|
|
7
|
+
# New captures are scrubbed on the way in (Services::EncodingSanitizer), but
|
|
8
|
+
# that does nothing for rows already in the table. SQLite and MySQL accept
|
|
9
|
+
# invalid UTF-8 and NUL bytes, and a row holding them makes the pages that
|
|
10
|
+
# render it answer 500. PostgreSQL rejects such rows at INSERT, so there is
|
|
11
|
+
# normally nothing to repair there; the scan is harmless.
|
|
12
|
+
#
|
|
13
|
+
# Behind `rake error_dashboard:scrub_invalid_encoding`. Safe to re-run: a
|
|
14
|
+
# second pass finds nothing to repair.
|
|
15
|
+
#
|
|
16
|
+
# @example
|
|
17
|
+
# ScrubInvalidEncoding.call # => { scanned: 1200, repaired: 3, unreadable: [] }
|
|
18
|
+
class ScrubInvalidEncoding
|
|
19
|
+
BATCH_SIZE = 500
|
|
20
|
+
|
|
21
|
+
MODEL_NAMES = %w[ErrorLog ErrorOccurrence].freeze
|
|
22
|
+
|
|
23
|
+
def self.call(batch_size: BATCH_SIZE)
|
|
24
|
+
new(batch_size: batch_size).call
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
def initialize(batch_size: BATCH_SIZE)
|
|
28
|
+
@batch_size = batch_size
|
|
29
|
+
@scanned = 0
|
|
30
|
+
@repaired = 0
|
|
31
|
+
@unreadable = []
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
# @return [Hash] { scanned:, repaired:, unreadable: [ "ErrorLog#12", ... ] }
|
|
35
|
+
def call
|
|
36
|
+
models.each { |model| scrub_model(model) }
|
|
37
|
+
|
|
38
|
+
{ scanned: @scanned, repaired: @repaired, unreadable: @unreadable }
|
|
39
|
+
end
|
|
40
|
+
|
|
41
|
+
private
|
|
42
|
+
|
|
43
|
+
def models
|
|
44
|
+
MODEL_NAMES.filter_map do |name|
|
|
45
|
+
model = RailsErrorDashboard.const_get(name)
|
|
46
|
+
model if model.table_exists?
|
|
47
|
+
rescue => e
|
|
48
|
+
RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] ScrubInvalidEncoding skipped #{name}: #{e.class}")
|
|
49
|
+
nil
|
|
50
|
+
end
|
|
51
|
+
end
|
|
52
|
+
|
|
53
|
+
def scrub_model(model)
|
|
54
|
+
columns = model.columns.select { |column| %i[string text].include?(column.type) }.map(&:name)
|
|
55
|
+
return if columns.empty?
|
|
56
|
+
|
|
57
|
+
model.in_batches(of: @batch_size) do |batch|
|
|
58
|
+
ids = batch.pluck(model.primary_key)
|
|
59
|
+
load_rows(model, ids).each { |row| scrub_row(row, columns) }
|
|
60
|
+
end
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
# One query per batch; if that batch cannot be materialised at all, fall
|
|
64
|
+
# back to row-by-row so a single unreadable row is reported by id instead
|
|
65
|
+
# of hiding the 499 readable ones around it.
|
|
66
|
+
def load_rows(model, ids)
|
|
67
|
+
model.where(model.primary_key => ids).to_a
|
|
68
|
+
rescue => e
|
|
69
|
+
RailsErrorDashboard::Logger.debug("[RailsErrorDashboard] ScrubInvalidEncoding batch read failed: #{e.class}")
|
|
70
|
+
ids.filter_map do |id|
|
|
71
|
+
model.find(id)
|
|
72
|
+
rescue => row_error
|
|
73
|
+
@unreadable << "#{model.name.demodulize}##{id} (#{row_error.class})"
|
|
74
|
+
nil
|
|
75
|
+
end
|
|
76
|
+
end
|
|
77
|
+
|
|
78
|
+
def scrub_row(row, columns)
|
|
79
|
+
@scanned += 1
|
|
80
|
+
|
|
81
|
+
changes = columns.each_with_object({}) do |column, fixed|
|
|
82
|
+
value = row.read_attribute(column)
|
|
83
|
+
next unless value.is_a?(String)
|
|
84
|
+
|
|
85
|
+
clean = Services::EncodingSanitizer.scrub(value)
|
|
86
|
+
fixed[column] = clean unless clean.equal?(value) || clean == value
|
|
87
|
+
end
|
|
88
|
+
# The model scrubs invalid strings as it loads a row, so by now they
|
|
89
|
+
# read as clean. It remembers which ones it had to repair.
|
|
90
|
+
row.invalid_encoding_attributes.each do |column|
|
|
91
|
+
changes[column] = row.read_attribute(column) if columns.include?(column)
|
|
92
|
+
end
|
|
93
|
+
return if changes.empty?
|
|
94
|
+
|
|
95
|
+
# update_columns: no callbacks, no validations, no updated_at bump. This
|
|
96
|
+
# is a byte-level repair, not an edit.
|
|
97
|
+
row.update_columns(changes)
|
|
98
|
+
@repaired += 1
|
|
99
|
+
rescue => e
|
|
100
|
+
@unreadable << "#{row.class.name.demodulize}##{row.id} (#{e.class})"
|
|
101
|
+
end
|
|
102
|
+
end
|
|
103
|
+
end
|
|
104
|
+
end
|