railwatch 0.1.4 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +212 -0
- data/README.md +10 -0
- data/app/channels/railwatch/environment_channel.rb +28 -0
- data/app/controllers/concerns/railwatch/telemetry_identity.rb +25 -0
- data/app/controllers/railwatch/alerts_controller.rb +48 -0
- data/app/controllers/railwatch/anomaly_rules_controller.rb +31 -0
- data/app/controllers/railwatch/attachments_controller.rb +22 -0
- data/app/controllers/railwatch/beacon_controller.rb +67 -7
- data/app/controllers/railwatch/broadcasts_controller.rb +15 -0
- data/app/controllers/railwatch/cache_events_controller.rb +30 -0
- data/app/controllers/railwatch/commands_controller.rb +16 -0
- data/app/controllers/railwatch/comments_controller.rb +11 -0
- data/app/controllers/railwatch/dashboard_controller.rb +69 -0
- data/app/controllers/railwatch/deploys_controller.rb +34 -0
- data/app/controllers/railwatch/deprecations_controller.rb +54 -0
- data/app/controllers/railwatch/environment_scoped.rb +99 -0
- data/app/controllers/railwatch/exceptions_controller.rb +53 -0
- data/app/controllers/railwatch/executions_controller.rb +13 -0
- data/app/controllers/railwatch/issues_controller.rb +258 -0
- data/app/controllers/railwatch/jobs_controller.rb +63 -0
- data/app/controllers/railwatch/llm_calls_controller.rb +124 -0
- data/app/controllers/railwatch/logs_controller.rb +40 -0
- data/app/controllers/railwatch/mails_controller.rb +19 -0
- data/app/controllers/railwatch/notifications_controller.rb +11 -0
- data/app/controllers/railwatch/outgoing_requests_controller.rb +19 -0
- data/app/controllers/railwatch/overview_controller.rb +38 -0
- data/app/controllers/railwatch/people_controller.rb +25 -0
- data/app/controllers/railwatch/processes_controller.rb +33 -0
- data/app/controllers/railwatch/profiles_controller.rb +50 -0
- data/app/controllers/railwatch/queries_controller.rb +59 -0
- data/app/controllers/railwatch/releases_controller.rb +75 -0
- data/app/controllers/railwatch/requests_controller.rb +55 -0
- data/app/controllers/railwatch/saved_views_controller.rb +53 -0
- data/app/controllers/railwatch/scheduled_tasks_controller.rb +45 -0
- data/app/controllers/railwatch/spans_controller.rb +47 -0
- data/app/controllers/railwatch/storage_ops_controller.rb +15 -0
- data/app/controllers/railwatch/tenants_controller.rb +44 -0
- data/app/controllers/railwatch/thresholds_controller.rb +40 -0
- data/app/controllers/railwatch/traces_controller.rb +72 -0
- data/app/controllers/railwatch/transactions_controller.rb +21 -0
- data/app/controllers/railwatch/view_renders_controller.rb +28 -0
- data/app/controllers/railwatch/visits_controller.rb +53 -0
- data/app/helpers/railwatch/assets_helper.rb +52 -0
- data/app/jobs/railwatch/anomaly_scan_job.rb +11 -0
- data/app/jobs/railwatch/application_job.rb +7 -0
- data/app/jobs/railwatch/auto_resolve_issues_job.rb +19 -0
- data/app/jobs/railwatch/check_scheduled_tasks_job.rb +113 -0
- data/app/jobs/railwatch/detect_anomalies_job.rb +171 -0
- data/app/jobs/railwatch/detect_performance_issues_job.rb +85 -0
- data/app/jobs/railwatch/group_exceptions_job.rb +91 -0
- data/app/jobs/railwatch/optimize_telemetry_job.rb +23 -0
- data/app/jobs/railwatch/performance_scan_job.rb +11 -0
- data/app/jobs/railwatch/prune_telemetry_job.rb +110 -0
- data/app/jobs/railwatch/release_health_rollup_job.rb +64 -0
- data/app/jobs/railwatch/rollup_catchup_job.rb +21 -0
- data/app/jobs/railwatch/rollup_job.rb +130 -0
- data/app/jobs/railwatch/scheduled_task_scan_job.rb +11 -0
- data/app/models/concerns/railwatch/detection_snapshotting.rb +20 -0
- data/app/models/railwatch/alert.rb +229 -0
- data/app/models/railwatch/alert_rule.rb +121 -0
- data/app/models/railwatch/anomaly_rule.rb +33 -0
- data/app/models/railwatch/application.rb +44 -0
- data/app/models/railwatch/application_record.rb +33 -0
- data/app/models/railwatch/comment.rb +36 -0
- data/app/models/railwatch/deploy.rb +67 -0
- data/app/models/railwatch/environment.rb +54 -0
- data/app/models/railwatch/execution_presenter.rb +185 -0
- data/app/models/railwatch/filter_query.rb +143 -0
- data/app/models/railwatch/followup_receipt.rb +27 -0
- data/app/models/railwatch/ingest/batch.rb +305 -0
- data/app/models/railwatch/ingest/mapper.rb +575 -0
- data/app/models/railwatch/ingest/payload.rb +96 -0
- data/app/models/railwatch/ingest/rollup_absorber.rb +170 -0
- data/app/models/railwatch/ingest/writer.rb +137 -0
- data/app/models/railwatch/issue.rb +277 -0
- data/app/models/railwatch/issue_activity.rb +23 -0
- data/app/models/railwatch/issue_detection_presenter.rb +309 -0
- data/app/models/railwatch/issue_detection_snapshot.rb +77 -0
- data/app/models/railwatch/maintenance_task.rb +53 -0
- data/app/models/railwatch/saved_view.rb +63 -0
- data/app/models/railwatch/telemetry/aggregations.rb +116 -0
- data/app/models/railwatch/telemetry/attachment.rb +31 -0
- data/app/models/railwatch/telemetry/bounded_gzip.rb +101 -0
- data/app/models/railwatch/telemetry/broadcast.rb +13 -0
- data/app/models/railwatch/telemetry/cache_event.rb +16 -0
- data/app/models/railwatch/telemetry/child.rb +53 -0
- data/app/models/railwatch/telemetry/cursor_page.rb +99 -0
- data/app/models/railwatch/telemetry/deprecation.rb +13 -0
- data/app/models/railwatch/telemetry/enqueued_job.rb +13 -0
- data/app/models/railwatch/telemetry/exception.rb +21 -0
- data/app/models/railwatch/telemetry/execution.rb +71 -0
- data/app/models/railwatch/telemetry/health_sample.rb +72 -0
- data/app/models/railwatch/telemetry/ingest_batch.rb +31 -0
- data/app/models/railwatch/telemetry/llm_call.rb +37 -0
- data/app/models/railwatch/telemetry/log.rb +85 -0
- data/app/models/railwatch/telemetry/mail.rb +13 -0
- data/app/models/railwatch/telemetry/n_plus_one.rb +92 -0
- data/app/models/railwatch/telemetry/notification.rb +13 -0
- data/app/models/railwatch/telemetry/outgoing_request.rb +13 -0
- data/app/models/railwatch/telemetry/person.rb +50 -0
- data/app/models/railwatch/telemetry/process.rb +10 -0
- data/app/models/railwatch/telemetry/profile.rb +37 -0
- data/app/models/railwatch/telemetry/query.rb +50 -0
- data/app/models/railwatch/telemetry/query_shape.rb +45 -0
- data/app/models/railwatch/telemetry/release_health.rb +63 -0
- data/app/models/railwatch/telemetry/rollup.rb +78 -0
- data/app/models/railwatch/telemetry/session.rb +24 -0
- data/app/models/railwatch/telemetry/span.rb +19 -0
- data/app/models/railwatch/telemetry/storage_op.rb +13 -0
- data/app/models/railwatch/telemetry/tenant.rb +221 -0
- data/app/models/railwatch/telemetry/transaction.rb +13 -0
- data/app/models/railwatch/telemetry/view_render.rb +13 -0
- data/app/models/railwatch/telemetry/visit.rb +24 -0
- data/app/models/railwatch/telemetry_record.rb +57 -0
- data/app/models/railwatch/threshold.rb +29 -0
- data/app/models/railwatch/user.rb +47 -0
- data/app/models/railwatch/viewer.rb +13 -0
- data/app/views/layouts/railwatch/dashboard.html.erb +25 -0
- data/config/routes.rb +59 -0
- data/db/railwatch_migrate/20260916000000_create_railwatch_tables.rb +151 -0
- data/db/railwatch_migrate/20260917000000_create_railwatch_maintenance_tasks.rb +18 -0
- data/db/railwatch_migrate/20260917120000_create_railwatch_followup_receipts.rb +20 -0
- data/db/railwatch_migrate/20260918120000_widen_host_user_ids.rb +71 -0
- data/db/railwatch_telemetry_migrate/20260903000001_create_telemetry.rb +481 -0
- data/db/railwatch_telemetry_migrate/20260903000002_rename_tenant_to_app_tenant.rb +14 -0
- data/db/railwatch_telemetry_migrate/20260903000003_add_statement_count_to_transactions.rb +9 -0
- data/db/railwatch_telemetry_migrate/20260903000004_add_role_and_channel.rb +10 -0
- data/db/railwatch_telemetry_migrate/20260903000005_add_locals_to_exceptions.rb +9 -0
- data/db/railwatch_telemetry_migrate/20260903000006_add_spans_health_vitals_and_fts.rb +82 -0
- data/db/railwatch_telemetry_migrate/20260903000007_rename_span_attributes_to_payload.rb +10 -0
- data/db/railwatch_telemetry_migrate/20260903000008_add_profiles_and_attachments.rb +60 -0
- data/db/railwatch_telemetry_migrate/20260903000009_add_truncated_to_attachments.rb +9 -0
- data/db/railwatch_telemetry_migrate/20260903000010_create_sessions_and_release_health.rb +51 -0
- data/db/railwatch_telemetry_migrate/20260903000011_add_fingerprint_to_exceptions.rb +11 -0
- data/db/railwatch_telemetry_migrate/20260904000012_add_failed_to_broadcasts.rb +10 -0
- data/db/railwatch_telemetry_migrate/20260904010000_add_filter_cursor_indexes.rb +37 -0
- data/db/railwatch_telemetry_migrate/20260904020000_add_n_plus_ones_execution_id_index.rb +11 -0
- data/db/railwatch_telemetry_migrate/20260904120000_create_query_shapes.rb +15 -0
- data/db/railwatch_telemetry_migrate/20260906120000_add_backpressure_factor_to_ingest_batches.rb +9 -0
- data/db/railwatch_telemetry_migrate/20260907000000_rename_lantern_version_on_processes.rb +10 -0
- data/db/railwatch_telemetry_migrate/20260913000000_rename_nightrail_version_on_processes.rb +9 -0
- data/db/railwatch_telemetry_migrate/20260914000000_drop_orphan_durable_ingest_tables.rb +18 -0
- data/db/railwatch_telemetry_migrate/20260914010000_drop_orphan_durable_ingest_columns.rb +25 -0
- data/db/railwatch_telemetry_migrate/20260915000000_create_llm_calls.rb +55 -0
- data/db/railwatch_telemetry_migrate/20260915120000_add_detail_to_llm_calls.rb +21 -0
- data/db/railwatch_telemetry_migrate/20260915200000_add_explained_index_to_queries.rb +16 -0
- data/db/railwatch_telemetry_migrate/20260915210000_add_slowest_index_to_queries.rb +18 -0
- data/db/railwatch_telemetry_migrate/20260917010000_add_batch_ledger_to_ingest_batches.rb +16 -0
- data/docs/configuration.md +26 -0
- data/docs/embedded.md +352 -0
- data/docs/getting-started.md +5 -0
- data/lib/generators/railwatch/install/install_generator.rb +222 -1
- data/lib/generators/railwatch/install/templates/{initializer.rb → initializer.rb.tt} +39 -0
- data/lib/generators/railwatch/install/templates/post-deploy +8 -0
- data/lib/puma/plugin/railwatch.rb +170 -0
- data/lib/railwatch/authentication.rb +83 -0
- data/lib/railwatch/configuration.rb +134 -3
- data/lib/railwatch/dashboard_assets.rb +45 -0
- data/lib/railwatch/embedded.rb +54 -0
- data/lib/railwatch/engine.rb +89 -1
- data/lib/railwatch/ingest_request_body_limit.rb +10 -0
- data/lib/railwatch/json_compat.rb +60 -0
- data/lib/railwatch/maintenance.rb +183 -0
- data/lib/railwatch/patches/runner_command.rb +21 -1
- data/lib/railwatch/record.rb +33 -8
- data/lib/railwatch/reporter.rb +16 -0
- data/lib/railwatch/subscribers/process_info.rb +2 -1
- data/lib/railwatch/transport/local.rb +78 -0
- data/lib/railwatch/transport/socket.rb +183 -0
- data/lib/railwatch/version.rb +1 -1
- data/lib/railwatch/writer.rb +370 -0
- data/lib/railwatch.rb +38 -1
- data/lib/tasks/railwatch_tasks.rake +128 -0
- data/public/railwatch/assets/CommitMono-Bold-D6h61ieg.woff2 +0 -0
- data/public/railwatch/assets/CommitMono-Regular-zr8w7Obm.woff2 +0 -0
- data/public/railwatch/assets/Roboto-Black-auA4GeOK.woff2 +0 -0
- data/public/railwatch/assets/Roboto-Bold-CJLnO8j1.woff2 +0 -0
- data/public/railwatch/assets/Roboto-Medium-Cm2bwKpj.woff2 +0 -0
- data/public/railwatch/assets/Roboto-Regular-Chaq1-PV.woff2 +0 -0
- data/public/railwatch/assets/app-layout-yh-sPWgK.js +1 -0
- data/public/railwatch/assets/app-wordmark-nbjkzxwQ.js +1 -0
- data/public/railwatch/assets/appearance-CDuRvQTB.js +1 -0
- data/public/railwatch/assets/application-C_kpBdnf.css +1 -0
- data/public/railwatch/assets/arrow-up-DVtOGdVA.js +1 -0
- data/public/railwatch/assets/auth-layout-TCwPpS1K.js +1 -0
- data/public/railwatch/assets/badge-Daw4Hvr8.js +1 -0
- data/public/railwatch/assets/braces-DhFbHPsz.js +1 -0
- data/public/railwatch/assets/card-BX_3HXcJ.js +1 -0
- data/public/railwatch/assets/chart-B14-N9g7.js +39 -0
- data/public/railwatch/assets/chart-hover-Vy52H4uD.js +1 -0
- data/public/railwatch/assets/chart-panel-Cb4S_ej_.js +1 -0
- data/public/railwatch/assets/checkbox-D3aRSsBj.js +1 -0
- data/public/railwatch/assets/code-KvW8k7Jr.js +1 -0
- data/public/railwatch/assets/copy-DaQMJWoT.js +1 -0
- data/public/railwatch/assets/copy-block-BKFGd11J.js +1 -0
- data/public/railwatch/assets/copy-id-Djks1fXB.js +1 -0
- data/public/railwatch/assets/cursor-load-more-BXV0f1D_.js +1 -0
- data/public/railwatch/assets/data-table-iv1bdF6u.js +1 -0
- data/public/railwatch/assets/edit-BZc_Iawe.js +1 -0
- data/public/railwatch/assets/edit-DMKUF8Zi.js +8 -0
- data/public/railwatch/assets/edit-__9yJlO3.js +1 -0
- data/public/railwatch/assets/empty-state-6j_0AaQQ.js +1 -0
- data/public/railwatch/assets/env-layout-DrT8rO6P.js +1 -0
- data/public/railwatch/assets/execution-path-CzgBUi5e.js +1 -0
- data/public/railwatch/assets/filter-bar-9SU5NrzX.js +1 -0
- data/public/railwatch/assets/flamegraph-qcekju8V.js +2 -0
- data/public/railwatch/assets/format-B9SDkrWj.js +1 -0
- data/public/railwatch/assets/frames-Cyu7KMxZ.js +1 -0
- data/public/railwatch/assets/google-sign-in-button-DsTSfmzY.js +1 -0
- data/public/railwatch/assets/index-1ol1-QWI.js +1 -0
- data/public/railwatch/assets/index-5jI4aFzC.js +1 -0
- data/public/railwatch/assets/index-9KTrVnrc.js +1 -0
- data/public/railwatch/assets/index-B0-8lcTp.js +1 -0
- data/public/railwatch/assets/index-B7jjfNfO.js +1 -0
- data/public/railwatch/assets/index-BBchRy0M.js +1 -0
- data/public/railwatch/assets/index-BRiq3SNR.js +1 -0
- data/public/railwatch/assets/index-BgKj9xhr.js +1 -0
- data/public/railwatch/assets/index-BkTZqqOu.js +1 -0
- data/public/railwatch/assets/index-BprKx8QO.js +1 -0
- data/public/railwatch/assets/index-C3jzvPs3.js +1 -0
- data/public/railwatch/assets/index-Cbs6gGyQ.js +1 -0
- data/public/railwatch/assets/index-CdRZ6AWF.js +1 -0
- data/public/railwatch/assets/index-CeYKnapu.js +1 -0
- data/public/railwatch/assets/index-Cmlwy1-V.js +1 -0
- data/public/railwatch/assets/index-Cwx6058d.js +1 -0
- data/public/railwatch/assets/index-D4CSdbHv.js +1 -0
- data/public/railwatch/assets/index-DEFMSkdG.js +1 -0
- data/public/railwatch/assets/index-DFiHEBSh.js +1 -0
- data/public/railwatch/assets/index-DL4vWdWJ.js +1 -0
- data/public/railwatch/assets/index-DU9F5b5d.js +1 -0
- data/public/railwatch/assets/index-DaXgPcGL.js +1 -0
- data/public/railwatch/assets/index-DbtaU-EE.js +1 -0
- data/public/railwatch/assets/index-DeOe83F4.js +1 -0
- data/public/railwatch/assets/index-DiucHN4B.js +1 -0
- data/public/railwatch/assets/index-DlnR_l9o.js +1 -0
- data/public/railwatch/assets/index-DlumCsWY.js +2 -0
- data/public/railwatch/assets/index-DmRd7aIG.js +1 -0
- data/public/railwatch/assets/index-DxSh2UpM.js +1 -0
- data/public/railwatch/assets/index-MIMGuFNt.js +1 -0
- data/public/railwatch/assets/index-OqI59zPb.js +1 -0
- data/public/railwatch/assets/index-P4rC7IlX.js +1 -0
- data/public/railwatch/assets/index-gpPOcFWq.js +1 -0
- data/public/railwatch/assets/index-oVkururr.js +1 -0
- data/public/railwatch/assets/index-p9puqVge.js +1 -0
- data/public/railwatch/assets/inertia-TViv6kNv.js +97 -0
- data/public/railwatch/assets/input-error-LxImUkxv.js +1 -0
- data/public/railwatch/assets/json-viewer-Ar4cjPDW.js +1 -0
- data/public/railwatch/assets/klass-CJ-J4INB.js +1 -0
- data/public/railwatch/assets/label-COUKWqE_.js +1 -0
- data/public/railwatch/assets/layout-DNSLAkw_.js +1 -0
- data/public/railwatch/assets/live-dot-ChfUtY3p.js +41 -0
- data/public/railwatch/assets/nav-CNnDqPlm.js +1 -0
- data/public/railwatch/assets/new-84S8ZJq9.js +1 -0
- data/public/railwatch/assets/new-Be55nmt9.js +1 -0
- data/public/railwatch/assets/new-Bi_xQiIb.js +1 -0
- data/public/railwatch/assets/new-BvCT8TMg.js +1 -0
- data/public/railwatch/assets/new-D05SajFR.js +1 -0
- data/public/railwatch/assets/new-D4uewYC8.js +1 -0
- data/public/railwatch/assets/onboarding-CYZi5Cqc.js +1 -0
- data/public/railwatch/assets/origin-identity-6q1-CBts.js +1 -0
- data/public/railwatch/assets/percentile-picker-gFZCXtdb.js +1 -0
- data/public/railwatch/assets/relative-time-IOOgl5n2.js +1 -0
- data/public/railwatch/assets/release-health-DC8oc7uw.js +1 -0
- data/public/railwatch/assets/route-Dv6LAWvT.js +1 -0
- data/public/railwatch/assets/segmented-h1VdDTqE.js +1 -0
- data/public/railwatch/assets/select-_AJsUa7X.js +1 -0
- data/public/railwatch/assets/separator-BwwTYtCF.js +1 -0
- data/public/railwatch/assets/series-chart-DaFPefku.js +1 -0
- data/public/railwatch/assets/show-B7NCgkEo.js +1 -0
- data/public/railwatch/assets/show-BKqyKjBK.js +1 -0
- data/public/railwatch/assets/show-BM6X2Mpo.js +1 -0
- data/public/railwatch/assets/show-BNw4tN5q.js +1 -0
- data/public/railwatch/assets/show-BO3bnG5h.js +1 -0
- data/public/railwatch/assets/show-BhrAVAEA.js +1 -0
- data/public/railwatch/assets/show-C4Ltf5i9.js +2 -0
- data/public/railwatch/assets/show-C8sHalnw.js +1 -0
- data/public/railwatch/assets/show-CeTL4B37.js +2 -0
- data/public/railwatch/assets/show-CpfgV1jP.js +1 -0
- data/public/railwatch/assets/show-DACku6AD.js +3 -0
- data/public/railwatch/assets/show-DIOSGcXV.js +6 -0
- data/public/railwatch/assets/show-DQp_1n-B.js +1 -0
- data/public/railwatch/assets/show-DVNz46RI.js +1 -0
- data/public/railwatch/assets/show-DYteoYWW.js +1 -0
- data/public/railwatch/assets/show-DgSIoRvA.js +1 -0
- data/public/railwatch/assets/show-JxFtB4eK.js +2 -0
- data/public/railwatch/assets/sort-header-DpFzXblu.js +1 -0
- data/public/railwatch/assets/source-link-B2183i2-.js +1 -0
- data/public/railwatch/assets/sparkline-cell-C3-5vFkP.js +1 -0
- data/public/railwatch/assets/stat-s4RpOS9w.js +1 -0
- data/public/railwatch/assets/status-badge-8jVV-LA4.js +1 -0
- data/public/railwatch/assets/tenant-path-G-6u9A-o.js +1 -0
- data/public/railwatch/assets/text-link-DfsiaCcP.js +1 -0
- data/public/railwatch/assets/textarea-Dye72uP7.js +1 -0
- data/public/railwatch/assets/timeline-CD7WHnbo.js +1 -0
- data/public/railwatch/assets/transition-B_AW8rMK.js +5 -0
- data/public/railwatch/assets/use-clipboard-ColgLyQ2.js +1 -0
- data/public/railwatch/icon.png +0 -0
- data/public/railwatch/icon.svg +5 -0
- data/public/railwatch/manifest.json +2171 -0
- data/public/railwatch/rails-vite.json +1 -0
- metadata +314 -4
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Railwatch
|
|
4
|
+
# Compares each enabled anomaly rule's current window against the same clock
|
|
5
|
+
# window on each of the previous `baseline_days` days and opens an anomaly
|
|
6
|
+
# issue when the current value sits above the baseline's mean plus
|
|
7
|
+
# `deviation` standard deviations. Runs on a schedule via AnomalyScanJob.
|
|
8
|
+
class DetectAnomaliesJob < ApplicationJob
|
|
9
|
+
include DetectionSnapshotting
|
|
10
|
+
|
|
11
|
+
queue_as :default
|
|
12
|
+
|
|
13
|
+
TYPE_FOR = DetectPerformanceIssuesJob::TYPE_FOR
|
|
14
|
+
# Fewer baseline days than this and the standard deviation is noise.
|
|
15
|
+
MIN_BASELINE_DAYS = AnomalyRule::MIN_BASELINE_DAYS
|
|
16
|
+
# Volume metrics on a handful of events swing wildly; ignore them.
|
|
17
|
+
MIN_EVENTS = 20
|
|
18
|
+
COOL_DOWN = 60.minutes
|
|
19
|
+
MAX_GROUPS_PER_RULE = 50
|
|
20
|
+
MAX_CURRENT_ROWS_PER_RULE = MAX_GROUPS_PER_RULE * 25
|
|
21
|
+
MAX_BASELINE_ROWS_PER_RULE = 50_000
|
|
22
|
+
|
|
23
|
+
def perform(environment)
|
|
24
|
+
now = Time.current
|
|
25
|
+
environment.anomaly_rules.enabled.find_each do |rule|
|
|
26
|
+
detect(environment, rule, now).each { |detection| open_issue(environment, rule, now, detection) }
|
|
27
|
+
end
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
private
|
|
31
|
+
|
|
32
|
+
def detect(environment, rule, now)
|
|
33
|
+
from = now - rule.window_minutes.minutes
|
|
34
|
+
type = TYPE_FOR.fetch(rule.target_kind)
|
|
35
|
+
environment.with_telemetry do
|
|
36
|
+
groups = current_groups(rule, type, from, now)
|
|
37
|
+
baseline = baseline_rows(rule, type, groups.keys, from, now)
|
|
38
|
+
return [] unless baseline
|
|
39
|
+
|
|
40
|
+
groups.filter_map do |group_hash, (name, summary)|
|
|
41
|
+
anomaly_for(rule, group_hash, name, summary, baseline.fetch(group_hash, []), from, now)
|
|
42
|
+
end
|
|
43
|
+
end
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
# Rollups are hourly, so a window shorter than an hour reads the whole
|
|
47
|
+
# enclosing hour bucket (Rollup.between rounds `from` down); the baseline
|
|
48
|
+
# windows are shifted by whole days and land on the same hour, so the
|
|
49
|
+
# comparison stays like-for-like.
|
|
50
|
+
def current_groups(rule, type, from, to)
|
|
51
|
+
scope = Telemetry::Rollup.for_type(type).between(from, to)
|
|
52
|
+
scope = scope.where(name: rule.target) unless rule.target == "*"
|
|
53
|
+
group_hashes = scope.reorder(nil).group(:group_hash).order(Arel.sql("SUM(count) DESC"), :group_hash)
|
|
54
|
+
.limit(MAX_GROUPS_PER_RULE + 1).pluck(:group_hash)
|
|
55
|
+
if group_hashes.length > MAX_GROUPS_PER_RULE
|
|
56
|
+
Rails.logger.warn("anomaly detector evaluated only the #{MAX_GROUPS_PER_RULE} busiest groups rule_id=#{rule.id}")
|
|
57
|
+
group_hashes.pop
|
|
58
|
+
end
|
|
59
|
+
rows = scope.where(group_hash: group_hashes).order(:bucket, :group_hash).limit(MAX_CURRENT_ROWS_PER_RULE + 1).to_a
|
|
60
|
+
if rows.length > MAX_CURRENT_ROWS_PER_RULE
|
|
61
|
+
Rails.logger.error("anomaly detector skipped rule_id=#{rule.id}: more than #{MAX_CURRENT_ROWS_PER_RULE} current rollups")
|
|
62
|
+
return {}
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
rows.group_by(&:group_hash)
|
|
66
|
+
.transform_values { |group_rows| [ group_rows.first.name, Telemetry::Rollup.summarize(group_rows) ] }
|
|
67
|
+
end
|
|
68
|
+
|
|
69
|
+
def anomaly_for(rule, group_hash, name, summary, baseline_rows, from, to)
|
|
70
|
+
return if cooling_down?(rule, group_hash)
|
|
71
|
+
current = value_for(rule.metric, summary, rule.window_minutes)
|
|
72
|
+
return if current.nil?
|
|
73
|
+
return if %w[count error_rate].include?(rule.metric) && summary[:count] < MIN_EVENTS
|
|
74
|
+
|
|
75
|
+
baseline = baseline_values(rule, baseline_rows, from, to)
|
|
76
|
+
return if baseline.size < MIN_BASELINE_DAYS
|
|
77
|
+
mean = baseline.sum / baseline.size
|
|
78
|
+
stddev = Math.sqrt(baseline.sum { |v| (v - mean)**2 } / baseline.size)
|
|
79
|
+
return unless current > mean + (rule.deviation * stddev) && current > mean * 1.25
|
|
80
|
+
|
|
81
|
+
# A perfectly flat baseline has no σ to measure against, so report the
|
|
82
|
+
# rule's own threshold rather than an infinite one.
|
|
83
|
+
sigmas = stddev.positive? ? (current - mean) / stddev : rule.deviation
|
|
84
|
+
{ group_hash: group_hash, name: name, current: current.round(2), mean: mean.round(2),
|
|
85
|
+
stddev: stddev.round(2), sigmas: sigmas.round(2) }
|
|
86
|
+
end
|
|
87
|
+
|
|
88
|
+
# The same clock window on each of the previous `baseline_days` days. Days
|
|
89
|
+
# with no rollups at all are left out rather than counted as zero.
|
|
90
|
+
def baseline_values(rule, rows, from, to)
|
|
91
|
+
(1..rule.baseline_days).filter_map do |days|
|
|
92
|
+
day_rows = rows.select { |row| row.bucket.between?((from - days.days).beginning_of_hour, to - days.days) }
|
|
93
|
+
next if day_rows.empty?
|
|
94
|
+
value_for(rule.metric, Telemetry::Rollup.summarize(day_rows), rule.window_minutes)
|
|
95
|
+
end
|
|
96
|
+
end
|
|
97
|
+
|
|
98
|
+
# One query for every group's baseline, reading only the hour buckets the
|
|
99
|
+
# baseline windows actually cover rather than the whole `baseline_days` span.
|
|
100
|
+
def baseline_rows(rule, type, group_hashes, from, to)
|
|
101
|
+
return {} if group_hashes.empty?
|
|
102
|
+
|
|
103
|
+
rows = Telemetry::Rollup.for_type(type)
|
|
104
|
+
.where(group_hash: group_hashes, bucket: baseline_buckets(rule, from, to))
|
|
105
|
+
.order(:bucket, :group_hash).limit(MAX_BASELINE_ROWS_PER_RULE + 1).to_a
|
|
106
|
+
if rows.length > MAX_BASELINE_ROWS_PER_RULE
|
|
107
|
+
Rails.logger.error("anomaly detector skipped rule_id=#{rule.id}: more than #{MAX_BASELINE_ROWS_PER_RULE} baseline rollups")
|
|
108
|
+
return nil
|
|
109
|
+
end
|
|
110
|
+
rows.group_by(&:group_hash)
|
|
111
|
+
end
|
|
112
|
+
|
|
113
|
+
# The hour buckets covered by the same clock window on each of the previous
|
|
114
|
+
# `baseline_days` days.
|
|
115
|
+
def baseline_buckets(rule, from, to)
|
|
116
|
+
(1..rule.baseline_days).flat_map do |days|
|
|
117
|
+
bucket = (from - days.days).beginning_of_hour
|
|
118
|
+
last = to - days.days
|
|
119
|
+
buckets = []
|
|
120
|
+
while bucket <= last
|
|
121
|
+
buckets << bucket
|
|
122
|
+
bucket += 1.hour
|
|
123
|
+
end
|
|
124
|
+
buckets
|
|
125
|
+
end
|
|
126
|
+
end
|
|
127
|
+
|
|
128
|
+
def value_for(metric, summary, window_minutes)
|
|
129
|
+
case metric
|
|
130
|
+
when "p95" then summary[:p95] / 1000.0
|
|
131
|
+
when "avg" then summary[:avg] / 1000.0
|
|
132
|
+
when "count" then summary[:count] / window_minutes.to_f
|
|
133
|
+
when "error_rate" then summary[:count].zero? ? nil : (summary[:errors] * 100.0 / summary[:count])
|
|
134
|
+
end
|
|
135
|
+
end
|
|
136
|
+
|
|
137
|
+
# Reads the primary database: one open anomaly issue per (rule, group) is
|
|
138
|
+
# enough for an hour, however often the scan runs.
|
|
139
|
+
def cooling_down?(rule, group_hash)
|
|
140
|
+
Issue.where(environment_id: rule.environment_id, group_hash: issue_group_hash(rule, group_hash))
|
|
141
|
+
.where(last_seen_at: COOL_DOWN.ago..).exists?
|
|
142
|
+
end
|
|
143
|
+
|
|
144
|
+
def issue_group_hash(rule, group_hash)
|
|
145
|
+
"anomaly:#{rule.id}:#{group_hash}"
|
|
146
|
+
end
|
|
147
|
+
|
|
148
|
+
def open_issue(environment, rule, now, detection)
|
|
149
|
+
issue, outcome = Issue.record_occurrence!(
|
|
150
|
+
environment: environment, group_hash: issue_group_hash(rule, detection[:group_hash]), kind: "anomaly",
|
|
151
|
+
title: "#{rule.metric} of #{detection[:name]} is #{detection[:sigmas].round(1)}σ above its #{rule.baseline_days}-day baseline (#{detection[:current]} vs #{detection[:mean]})",
|
|
152
|
+
culprit: detection[:name], occurred_at: now, deploy: nil, user_ref: nil,
|
|
153
|
+
sample: { rule_id: rule.id, metric: rule.metric, current: detection[:current], mean: detection[:mean],
|
|
154
|
+
stddev: detection[:stddev], sigmas: detection[:sigmas], window_minutes: rule.window_minutes })
|
|
155
|
+
carry_detection(issue, outcome) do
|
|
156
|
+
IssueDetectionSnapshot.capture(
|
|
157
|
+
environment: environment, kind: "anomaly", telemetry_type: TYPE_FOR.fetch(rule.target_kind),
|
|
158
|
+
telemetry_group_hash: detection[:group_hash], target: detection[:name], metric: rule.metric,
|
|
159
|
+
from: now - rule.window_minutes.minutes, to: now, value: detection[:current],
|
|
160
|
+
baseline: detection.slice(:mean, :stddev, :sigmas).merge(days: rule.baseline_days, deviation: rule.deviation),
|
|
161
|
+
rule: { type: "anomaly", id: rule.id, target_kind: rule.target_kind, target: rule.target }
|
|
162
|
+
)
|
|
163
|
+
end
|
|
164
|
+
if outcome == :new || outcome == :regressed
|
|
165
|
+
issue.fire_alerts!("anomaly", rule_id: rule.id, metric: rule.metric, current: detection[:current],
|
|
166
|
+
mean: detection[:mean], sigmas: detection[:sigmas])
|
|
167
|
+
end
|
|
168
|
+
rule.update_columns(last_fired_at: now)
|
|
169
|
+
end
|
|
170
|
+
end
|
|
171
|
+
end
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Railwatch
|
|
4
|
+
# Evaluates every threshold for an environment against the last window of
|
|
5
|
+
# rollups and opens performance issues. Runs on a schedule (recurring.yml).
|
|
6
|
+
class DetectPerformanceIssuesJob < ApplicationJob
|
|
7
|
+
include DetectionSnapshotting
|
|
8
|
+
|
|
9
|
+
queue_as :default
|
|
10
|
+
|
|
11
|
+
MAX_GROUPS_PER_RULE = 200
|
|
12
|
+
MAX_ROLLUP_ROWS_PER_RULE = MAX_GROUPS_PER_RULE * 25
|
|
13
|
+
|
|
14
|
+
TYPE_FOR = { "requests" => "request", "jobs" => "job_attempt", "commands" => "command", "queries" => "query",
|
|
15
|
+
"scheduled_tasks" => "scheduled_task", "outgoing_requests" => "outgoing_request" }.freeze
|
|
16
|
+
|
|
17
|
+
def perform(environment)
|
|
18
|
+
now = Time.current
|
|
19
|
+
environment.thresholds.find_each do |threshold|
|
|
20
|
+
from = now - threshold.window_minutes.minutes
|
|
21
|
+
type = TYPE_FOR.fetch(threshold.target_kind)
|
|
22
|
+
groups = environment.with_telemetry do
|
|
23
|
+
scope = Telemetry::Rollup.for_type(type).between(from, now)
|
|
24
|
+
scope = scope.where(name: threshold.target) unless threshold.target == "*"
|
|
25
|
+
bounded_groups(scope, threshold)
|
|
26
|
+
end
|
|
27
|
+
groups.each do |group_hash, (name, summary)|
|
|
28
|
+
value = value_for(threshold.metric, summary)
|
|
29
|
+
next if value.nil? || value <= threshold.limit
|
|
30
|
+
open_issue(environment, threshold, group_hash, name, value, from, now)
|
|
31
|
+
end
|
|
32
|
+
end
|
|
33
|
+
end
|
|
34
|
+
|
|
35
|
+
private
|
|
36
|
+
|
|
37
|
+
def bounded_groups(scope, threshold)
|
|
38
|
+
group_hashes = scope.reorder(nil).group(:group_hash).order(Arel.sql("SUM(count) DESC"), :group_hash)
|
|
39
|
+
.limit(MAX_GROUPS_PER_RULE + 1).pluck(:group_hash)
|
|
40
|
+
if group_hashes.length > MAX_GROUPS_PER_RULE
|
|
41
|
+
Rails.logger.warn("threshold detector evaluated only the #{MAX_GROUPS_PER_RULE} busiest groups threshold_id=#{threshold.id}")
|
|
42
|
+
group_hashes.pop
|
|
43
|
+
end
|
|
44
|
+
rows = scope.where(group_hash: group_hashes).order(:bucket, :group_hash).limit(MAX_ROLLUP_ROWS_PER_RULE + 1).to_a
|
|
45
|
+
if rows.length > MAX_ROLLUP_ROWS_PER_RULE
|
|
46
|
+
Rails.logger.error("threshold detector skipped threshold_id=#{threshold.id}: more than #{MAX_ROLLUP_ROWS_PER_RULE} rollups")
|
|
47
|
+
return {}
|
|
48
|
+
end
|
|
49
|
+
|
|
50
|
+
rows.group_by(&:group_hash)
|
|
51
|
+
.transform_values { |group_rows| [ group_rows.first.name, Telemetry::Rollup.summarize(group_rows) ] }
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
def value_for(metric, s)
|
|
55
|
+
case metric
|
|
56
|
+
when "p95" then s[:p95] / 1000.0
|
|
57
|
+
when "max" then s[:max] / 1000.0
|
|
58
|
+
when "avg" then s[:avg] / 1000.0
|
|
59
|
+
when "error_rate", "failure_rate" then s[:count].zero? ? nil : (s[:errors] * 100.0 / s[:count])
|
|
60
|
+
end
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
def open_issue(environment, threshold, group_hash, name, value, from, now)
|
|
64
|
+
issue, outcome = Issue.record_occurrence!(
|
|
65
|
+
environment: environment, group_hash: "perf:#{threshold.id}:#{group_hash}", kind: "performance",
|
|
66
|
+
title: "#{name} exceeded #{threshold.metric} #{threshold.limit}#{threshold.metric.end_with?('rate') ? '%' : 'ms'} (#{value.round(1)})",
|
|
67
|
+
culprit: name, occurred_at: now, deploy: nil, user_ref: nil,
|
|
68
|
+
sample: { threshold_id: threshold.id, value: value.round(2), metric: threshold.metric, limit: threshold.limit })
|
|
69
|
+
carry_detection(issue, outcome) do
|
|
70
|
+
IssueDetectionSnapshot.capture(
|
|
71
|
+
environment: environment, kind: "performance", telemetry_type: TYPE_FOR.fetch(threshold.target_kind),
|
|
72
|
+
telemetry_group_hash: group_hash, target: name, metric: threshold.metric, from: from, to: now,
|
|
73
|
+
value: value.round(2), limit: threshold.limit,
|
|
74
|
+
rule: { type: "threshold", id: threshold.id, target_kind: threshold.target_kind, target: threshold.target }
|
|
75
|
+
)
|
|
76
|
+
end
|
|
77
|
+
return unless outcome == :new || outcome == :regressed
|
|
78
|
+
payload = { issue_key: issue.key, title: issue.title, environment: environment.name }
|
|
79
|
+
environment.application.alert_rules.where(event: "threshold").find_each do |rule|
|
|
80
|
+
next unless rule.matches?(issue: issue, payload: payload)
|
|
81
|
+
rule.fire!(event: "threshold", issue: issue, payload: payload)
|
|
82
|
+
end
|
|
83
|
+
end
|
|
84
|
+
end
|
|
85
|
+
end
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Railwatch
|
|
4
|
+
# Turns newly ingested exception rows into Issues (create or bump), and
|
|
5
|
+
# fires new / regressed alerts.
|
|
6
|
+
class GroupExceptionsJob < ApplicationJob
|
|
7
|
+
queue_as :default
|
|
8
|
+
|
|
9
|
+
# A browser whose in-flight request died while kamal-proxy swapped
|
|
10
|
+
# containers reports a network error a few seconds after the deploy
|
|
11
|
+
# marker. That is the deploy working, not the app failing, and it opened
|
|
12
|
+
# (then regressed) an issue on every deploy today. The exception rows are
|
|
13
|
+
# still recorded; they just do not become an issue when they land inside
|
|
14
|
+
# this window around a deploy. Same idea as CheckScheduledTasksJob's
|
|
15
|
+
# DEPLOY_GRACE for missed runs.
|
|
16
|
+
DEPLOY_GRACE = 3.minutes
|
|
17
|
+
DEPLOY_NETWORK_ERRORS = %w[HttpNetworkError AxiosError InertiaException NetworkError TypeError:NetworkError].freeze
|
|
18
|
+
|
|
19
|
+
# batch_id: the embedded ledger's id for the batch these rows came from.
|
|
20
|
+
# With one, each group's occurrence count is committed together with a
|
|
21
|
+
# FollowupReceipt for (batch, group), so running this again for the same
|
|
22
|
+
# batch (a crash after the count but before the ledger was cleared, or
|
|
23
|
+
# the inline drain racing the maintenance drain) counts nothing twice.
|
|
24
|
+
# The hosted platform enqueues this once per HTTP batch and passes none.
|
|
25
|
+
def perform(environment, exception_ids, batch_id: nil)
|
|
26
|
+
rows = environment.with_telemetry { Telemetry::Exception.where(id: exception_ids).to_a }
|
|
27
|
+
rows = rows.reject { |row| deploy_swap_noise?(environment, row) }
|
|
28
|
+
rows.group_by(&:group_hash).each do |group_hash, group|
|
|
29
|
+
issue = outcome = nil
|
|
30
|
+
Issue.transaction do
|
|
31
|
+
next if batch_id && !FollowupReceipt.claim!(batch_id: batch_id, group_hash: group_hash)
|
|
32
|
+
|
|
33
|
+
latest = group.max_by(&:occurred_at)
|
|
34
|
+
issue, outcome = Issue.record_occurrence!(
|
|
35
|
+
environment: environment, group_hash: group_hash, kind: "exception",
|
|
36
|
+
title: "#{latest.class_name}: #{latest.message.to_s.first(200)}",
|
|
37
|
+
culprit: [ latest.file, latest.line ].compact.join(":").presence,
|
|
38
|
+
occurred_at: latest.occurred_at, deploy: latest.deploy, user_ref: latest.user_ref, source: latest.source,
|
|
39
|
+
sample: { exception_id: latest.id, handled: latest.handled, execution_id: latest.execution_id,
|
|
40
|
+
execution_preview: latest.execution_preview, deploy: latest.deploy,
|
|
41
|
+
fingerprint: latest.fingerprint, fingerprint_source: latest.fingerprint_source })
|
|
42
|
+
issue.increment!(:occurrences, group.size - 1) if group.size > 1
|
|
43
|
+
end
|
|
44
|
+
next unless issue
|
|
45
|
+
|
|
46
|
+
update_affected_users(environment, issue) if affected_users_due?(issue, outcome)
|
|
47
|
+
alert(issue, outcome)
|
|
48
|
+
end
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
# The affected-user count is a DISTINCT over every retained occurrence of
|
|
52
|
+
# the group, so on a busy issue it grows with the retention window and
|
|
53
|
+
# used to run once per batch per group. Once per window per issue, and
|
|
54
|
+
# always for a new issue, keeps the number fresh at a bounded cost.
|
|
55
|
+
AFFECTED_USERS_WINDOW = 5.minutes
|
|
56
|
+
AFFECTED_USERS_AT = Concurrent::Map.new
|
|
57
|
+
|
|
58
|
+
private
|
|
59
|
+
|
|
60
|
+
def affected_users_due?(issue, outcome)
|
|
61
|
+
now = Time.current
|
|
62
|
+
return AFFECTED_USERS_AT[issue.id] = now if outcome == :new
|
|
63
|
+
|
|
64
|
+
last = AFFECTED_USERS_AT[issue.id]
|
|
65
|
+
return false if last && now - last < AFFECTED_USERS_WINDOW
|
|
66
|
+
|
|
67
|
+
AFFECTED_USERS_AT[issue.id] = now
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
def deploy_swap_noise?(environment, row)
|
|
71
|
+
return false unless row.source == "browser" && DEPLOY_NETWORK_ERRORS.include?(row.class_name)
|
|
72
|
+
|
|
73
|
+
last_deploy_at = environment.deploys.maximum(:deployed_at) or return false
|
|
74
|
+
row.occurred_at.between?(last_deploy_at - DEPLOY_GRACE, last_deploy_at + DEPLOY_GRACE)
|
|
75
|
+
end
|
|
76
|
+
|
|
77
|
+
def update_affected_users(environment, issue)
|
|
78
|
+
count = environment.with_telemetry { Telemetry::Exception.where(group_hash: issue.group_hash).where.not(user_ref: nil).distinct.count(:user_ref) }
|
|
79
|
+
issue.update_columns(affected_users: count)
|
|
80
|
+
end
|
|
81
|
+
|
|
82
|
+
def alert(issue, outcome)
|
|
83
|
+
event = { new: "new_issue", regressed: "regressed_issue" }[outcome] or return
|
|
84
|
+
payload = { issue_key: issue.key, title: issue.title, environment: issue.environment.name }
|
|
85
|
+
issue.application.alert_rules.where(event: event).find_each do |rule|
|
|
86
|
+
next unless rule.matches?(issue: issue, payload: payload)
|
|
87
|
+
rule.fire!(event: event, issue: issue, payload: payload)
|
|
88
|
+
end
|
|
89
|
+
end
|
|
90
|
+
end
|
|
91
|
+
end
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Railwatch
|
|
4
|
+
# Refreshes SQLite's planner statistics for each environment's telemetry
|
|
5
|
+
# database. Nothing else ever runs ANALYZE on these files, so the planner
|
|
6
|
+
# has been choosing index shapes blind on tables that grow by hundreds of
|
|
7
|
+
# thousands of rows a day. PRAGMA optimize only re-analyzes tables whose
|
|
8
|
+
# stats look stale, so it is cheap to run daily; analysis_limit bounds the
|
|
9
|
+
# rows it samples per index.
|
|
10
|
+
class OptimizeTelemetryJob < ApplicationJob
|
|
11
|
+
queue_as :maintenance
|
|
12
|
+
|
|
13
|
+
def perform(environment = nil)
|
|
14
|
+
return [ Environment.current ].each { |env| self.class.perform_later(env) } if environment.nil?
|
|
15
|
+
|
|
16
|
+
environment.with_telemetry do
|
|
17
|
+
connection = TelemetryRecord.connection
|
|
18
|
+
connection.execute("PRAGMA analysis_limit = 1000")
|
|
19
|
+
connection.execute("PRAGMA optimize")
|
|
20
|
+
end
|
|
21
|
+
end
|
|
22
|
+
end
|
|
23
|
+
end
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Railwatch
|
|
4
|
+
# Deletes raw rows past the account's retention window. Rollups are kept
|
|
5
|
+
# for 13 months so long-range charts keep working after raw rows are gone.
|
|
6
|
+
# Telemetry::Session is monitored-application telemetry; dashboard login
|
|
7
|
+
# sessions live in the primary database and are deliberately not touched.
|
|
8
|
+
class PruneTelemetryJob < ApplicationJob
|
|
9
|
+
queue_as :maintenance
|
|
10
|
+
|
|
11
|
+
AGGREGATE_RETENTION = 13.months
|
|
12
|
+
# Sessions have never been pruned, so the first run of this job would walk
|
|
13
|
+
# every hour an environment has ever recorded. Delete a bounded number of
|
|
14
|
+
# hourly buckets per run and let the nightly schedule drain the backlog.
|
|
15
|
+
SESSION_HOURS_PER_RUN = 48
|
|
16
|
+
BATCH = 5_000
|
|
17
|
+
# Per raw table per run. A backlog past this (a retention change, a
|
|
18
|
+
# restore of an old file) drains over successive runs rather than in one
|
|
19
|
+
# that holds SQLite's write lock for as long as it takes. In the embedded
|
|
20
|
+
# writer that long hold would look like a wedge and end the process.
|
|
21
|
+
MAX_BATCHES_PER_TABLE = 40
|
|
22
|
+
|
|
23
|
+
RAW = [ Telemetry::Execution, Telemetry::Query, Telemetry::Exception, Telemetry::CacheEvent, Telemetry::Mail,
|
|
24
|
+
Telemetry::Broadcast, Telemetry::Notification, Telemetry::OutgoingRequest, Telemetry::StorageOp,
|
|
25
|
+
Telemetry::ViewRender, Telemetry::Log, Telemetry::EnqueuedJob, Telemetry::Transaction,
|
|
26
|
+
Telemetry::NPlusOne, Telemetry::Deprecation, Telemetry::Visit, Telemetry::Span,
|
|
27
|
+
Telemetry::LlmCall,
|
|
28
|
+
Telemetry::Profile, Telemetry::Attachment ].freeze
|
|
29
|
+
|
|
30
|
+
# checkpoint: the WAL checkpoint mode run at the end. TRUNCATE (the
|
|
31
|
+
# default, for a dedicated worker) hands the space back to the filesystem
|
|
32
|
+
# but blocks every reader and writer while it does; PASSIVE checkpoints
|
|
33
|
+
# what it can without waiting on anyone, which is what a prune running
|
|
34
|
+
# inside a Puma worker (Railwatch::Maintenance) must use.
|
|
35
|
+
def perform(environment = nil, checkpoint: "TRUNCATE")
|
|
36
|
+
return [ Environment.current ].each { |env| self.class.perform_later(env) } if environment.nil?
|
|
37
|
+
|
|
38
|
+
cutoff = environment.retention_days.days.ago
|
|
39
|
+
aggregate_cutoff = AGGREGATE_RETENTION.ago
|
|
40
|
+
environment.with_telemetry do
|
|
41
|
+
unindex_logs(cutoff) if Telemetry::Log.fts_available?
|
|
42
|
+
RAW.each { |klass| prune(klass, cutoff) }
|
|
43
|
+
# NOT EXISTS rather than NOT IN: queries.group_hash is nullable, and one
|
|
44
|
+
# NULL in a NOT IN subquery makes it match nothing. A few hundred shapes.
|
|
45
|
+
Telemetry::QueryShape.where("NOT EXISTS (SELECT 1 FROM queries WHERE queries.group_hash = query_shapes.group_hash)").delete_all
|
|
46
|
+
prune_sessions(cutoff)
|
|
47
|
+
Telemetry::Rollup.where(bucket: ...aggregate_cutoff).delete_all
|
|
48
|
+
Telemetry::ReleaseHealth.where(bucket: ...aggregate_cutoff).in_batches(of: 5_000).delete_all
|
|
49
|
+
Telemetry::Person.where(last_seen_at: ...cutoff).in_batches(of: 5_000).delete_all
|
|
50
|
+
Telemetry::IngestBatch.where(received_at: ...cutoff).delete_all
|
|
51
|
+
Telemetry::Process.where(booted_at: ...cutoff).delete_all
|
|
52
|
+
Telemetry::HealthSample.where(sampled_at: ...cutoff).delete_all
|
|
53
|
+
TelemetryRecord.connection.execute("PRAGMA wal_checkpoint(#{checkpoint})") if TelemetryRecord.connection.adapter_name =~ /sqlite/i
|
|
54
|
+
end
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
private
|
|
58
|
+
|
|
59
|
+
# in_batches walks a table by primary key: its probe is "WHERE occurred_at
|
|
60
|
+
# < ? ORDER BY id LIMIT n", which no index serves, so SQLite scanned the
|
|
61
|
+
# whole table by rowid to find nothing to delete -- 42 seconds on the
|
|
62
|
+
# platform's own 12M-row queries table, every night, for an empty result.
|
|
63
|
+
# Ordered by occurred_at the same probe is one seek on any index led by
|
|
64
|
+
# that column, and the first page of expired rows is exactly the oldest
|
|
65
|
+
# ones. Tables without such an index still scan, but they are the small
|
|
66
|
+
# ones.
|
|
67
|
+
def prune(klass, cutoff)
|
|
68
|
+
MAX_BATCHES_PER_TABLE.times do
|
|
69
|
+
# Ordered by (occurred_at, id), not occurred_at alone: rows that tie
|
|
70
|
+
# on the timestamp at the limit boundary would otherwise be a
|
|
71
|
+
# different set here than in unindex_logs above, which leaves stale
|
|
72
|
+
# FTS postings behind and withdraws postings for logs that are staying.
|
|
73
|
+
ids = klass.where(occurred_at: ...cutoff).order(:occurred_at, :id).limit(BATCH).pluck(:id)
|
|
74
|
+
break if ids.empty?
|
|
75
|
+
klass.where(id: ids).delete_all
|
|
76
|
+
break if ids.size < BATCH
|
|
77
|
+
end
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
# Sessions are the raw input to the hourly release_health aggregate, so they
|
|
81
|
+
# are deleted a whole hour at a time. Flooring the cutoff to the start of its
|
|
82
|
+
# hour leaves the partial hour straddling the boundary in place: deleting
|
|
83
|
+
# only its expired prefix would let the next rollup rebuild that hour from
|
|
84
|
+
# the surviving tail and silently shrink an already correct aggregate.
|
|
85
|
+
def prune_sessions(cutoff)
|
|
86
|
+
expired = Telemetry::Session.where(occurred_at: ...cutoff.utc.beginning_of_hour)
|
|
87
|
+
SESSION_HOURS_PER_RUN.times do
|
|
88
|
+
oldest = expired.minimum(:occurred_at)
|
|
89
|
+
break if oldest.nil?
|
|
90
|
+
|
|
91
|
+
hour = oldest.utc.beginning_of_hour
|
|
92
|
+
expired.where(occurred_at: hour...(hour + 1.hour)).in_batches(of: 5_000).delete_all
|
|
93
|
+
end
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
# logs_fts is an external-content index with no triggers, so the postings
|
|
97
|
+
# for a log line have to be withdrawn while its row (and message) is still
|
|
98
|
+
# there. Deleting the rows first would leave the index permanently out of
|
|
99
|
+
# sync -- searches would keep returning rowids that no longer exist.
|
|
100
|
+
# Bounded the same way as the rows it precedes: the postings for at most
|
|
101
|
+
# MAX_BATCHES_PER_TABLE * BATCH expired lines are withdrawn per run, which
|
|
102
|
+
# is exactly the set prune(Telemetry::Log) will delete this run.
|
|
103
|
+
def unindex_logs(cutoff)
|
|
104
|
+
Telemetry::Log.connection.execute(Telemetry::Log.sanitize_sql_array([
|
|
105
|
+
"INSERT INTO logs_fts(logs_fts, rowid, message) SELECT 'delete', id, message FROM logs " \
|
|
106
|
+
"WHERE occurred_at < ? ORDER BY occurred_at, id LIMIT ?", cutoff, MAX_BATCHES_PER_TABLE * BATCH
|
|
107
|
+
]))
|
|
108
|
+
end
|
|
109
|
+
end
|
|
110
|
+
end
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Railwatch
|
|
4
|
+
# Recomputes one hourly release_health bucket from raw `sessions` rows.
|
|
5
|
+
# Idempotent, and debounced by the same Solid Queue concurrency control
|
|
6
|
+
# RollupJob uses: while one recompute of a bucket is queued or running,
|
|
7
|
+
# further enqueues for it are discarded rather than piling up.
|
|
8
|
+
#
|
|
9
|
+
# The gem re-sends a live session every flush interval, so the same
|
|
10
|
+
# session_id lands in a bucket many times. Each id collapses to the worst
|
|
11
|
+
# status it reached (crashed beats errored beats ok beats started) and its
|
|
12
|
+
# longest reported duration, which is what makes "sessions" a session count
|
|
13
|
+
# rather than a record count.
|
|
14
|
+
class ReleaseHealthRollupJob < ApplicationJob
|
|
15
|
+
queue_as :rollups
|
|
16
|
+
limits_concurrency to: 1, key: ->(environment, bucket) { "release_health:#{environment.id}:#{bucket.to_i}" }, duration: 10.minutes, on_conflict: :discard
|
|
17
|
+
|
|
18
|
+
RANKS = { "started" => 0, "ok" => 1, "errored" => 2, "crashed" => 3 }.freeze
|
|
19
|
+
CRASHED = RANKS["crashed"]
|
|
20
|
+
ERRORED = RANKS["errored"]
|
|
21
|
+
|
|
22
|
+
def perform(environment, bucket)
|
|
23
|
+
bucket = bucket.utc.beginning_of_hour
|
|
24
|
+
# PruneTelemetryJob deletes raw sessions in whole hours below this floor.
|
|
25
|
+
# A rollup enqueued before the prune can run after it, and rebuilding a
|
|
26
|
+
# pruned hour would replace a complete aggregate with an empty one.
|
|
27
|
+
return if bucket < environment.retention_days.days.ago.utc.beginning_of_hour
|
|
28
|
+
|
|
29
|
+
environment.with_telemetry do
|
|
30
|
+
rows = Telemetry::Session.where(occurred_at: bucket...(bucket + 1.hour))
|
|
31
|
+
.where.not(deploy: nil).pluck(:deploy, :session_id, :status, :duration, :user_ref)
|
|
32
|
+
aggregates = rows.group_by(&:first).map { |deploy, group| aggregate(deploy, bucket, collapse(group)) }
|
|
33
|
+
Telemetry::ReleaseHealth.transaction do
|
|
34
|
+
Telemetry::ReleaseHealth.where(bucket: bucket).delete_all
|
|
35
|
+
Telemetry::ReleaseHealth.insert_all(aggregates) if aggregates.any?
|
|
36
|
+
end
|
|
37
|
+
end
|
|
38
|
+
end
|
|
39
|
+
|
|
40
|
+
private
|
|
41
|
+
|
|
42
|
+
# session_id => the worst status, longest duration, and first user seen for
|
|
43
|
+
# it in this bucket.
|
|
44
|
+
def collapse(rows)
|
|
45
|
+
rows.each_with_object({}) do |(_deploy, session_id, status, duration, user_ref), sessions|
|
|
46
|
+
session = sessions[session_id] ||= { rank: 0, duration: nil, user: nil }
|
|
47
|
+
session[:rank] = [ session[:rank], RANKS.fetch(status.to_s, 0) ].max
|
|
48
|
+
session[:duration] = [ session[:duration] || 0, duration ].max if duration
|
|
49
|
+
session[:user] ||= user_ref
|
|
50
|
+
end
|
|
51
|
+
end
|
|
52
|
+
|
|
53
|
+
def aggregate(deploy, bucket, sessions)
|
|
54
|
+
durations = sessions.values.filter_map { |s| s[:duration] }
|
|
55
|
+
users = sessions.values.filter_map { |s| s[:user] }.uniq
|
|
56
|
+
crashed_users = sessions.values.select { |s| s[:rank] == CRASHED }.filter_map { |s| s[:user] }.uniq
|
|
57
|
+
{ deploy: deploy, bucket: bucket, sessions: sessions.size,
|
|
58
|
+
sessions_errored: sessions.values.count { |s| s[:rank] == ERRORED },
|
|
59
|
+
sessions_crashed: sessions.values.count { |s| s[:rank] == CRASHED },
|
|
60
|
+
users: users.size, users_crashed: crashed_users.size,
|
|
61
|
+
duration_sum: durations.sum, duration_count: durations.size }
|
|
62
|
+
end
|
|
63
|
+
end
|
|
64
|
+
end
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Railwatch
|
|
4
|
+
# Recomputes the current and previous hour so rollups are never more than a
|
|
5
|
+
# schedule tick stale, whatever happened to the per-batch RollupJob enqueues
|
|
6
|
+
# (debounced in the web process, and their concurrency semaphore can outlive
|
|
7
|
+
# a worker restart). The platform iterates every active environment; an
|
|
8
|
+
# embedded install has one.
|
|
9
|
+
class RollupCatchupJob < ApplicationJob
|
|
10
|
+
queue_as :rollups
|
|
11
|
+
|
|
12
|
+
def perform
|
|
13
|
+
env = Environment.current
|
|
14
|
+
now = Time.current
|
|
15
|
+
[ now.beginning_of_hour, (now - 1.hour).beginning_of_hour ].each do |bucket|
|
|
16
|
+
RollupJob.perform_now(env, bucket)
|
|
17
|
+
ReleaseHealthRollupJob.perform_now(env, bucket)
|
|
18
|
+
end
|
|
19
|
+
end
|
|
20
|
+
end
|
|
21
|
+
end
|