railwatch 0.1.4 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +212 -0
- data/README.md +10 -0
- data/app/channels/railwatch/environment_channel.rb +28 -0
- data/app/controllers/concerns/railwatch/telemetry_identity.rb +25 -0
- data/app/controllers/railwatch/alerts_controller.rb +48 -0
- data/app/controllers/railwatch/anomaly_rules_controller.rb +31 -0
- data/app/controllers/railwatch/attachments_controller.rb +22 -0
- data/app/controllers/railwatch/beacon_controller.rb +67 -7
- data/app/controllers/railwatch/broadcasts_controller.rb +15 -0
- data/app/controllers/railwatch/cache_events_controller.rb +30 -0
- data/app/controllers/railwatch/commands_controller.rb +16 -0
- data/app/controllers/railwatch/comments_controller.rb +11 -0
- data/app/controllers/railwatch/dashboard_controller.rb +69 -0
- data/app/controllers/railwatch/deploys_controller.rb +34 -0
- data/app/controllers/railwatch/deprecations_controller.rb +54 -0
- data/app/controllers/railwatch/environment_scoped.rb +99 -0
- data/app/controllers/railwatch/exceptions_controller.rb +53 -0
- data/app/controllers/railwatch/executions_controller.rb +13 -0
- data/app/controllers/railwatch/issues_controller.rb +258 -0
- data/app/controllers/railwatch/jobs_controller.rb +63 -0
- data/app/controllers/railwatch/llm_calls_controller.rb +124 -0
- data/app/controllers/railwatch/logs_controller.rb +40 -0
- data/app/controllers/railwatch/mails_controller.rb +19 -0
- data/app/controllers/railwatch/notifications_controller.rb +11 -0
- data/app/controllers/railwatch/outgoing_requests_controller.rb +19 -0
- data/app/controllers/railwatch/overview_controller.rb +38 -0
- data/app/controllers/railwatch/people_controller.rb +25 -0
- data/app/controllers/railwatch/processes_controller.rb +33 -0
- data/app/controllers/railwatch/profiles_controller.rb +50 -0
- data/app/controllers/railwatch/queries_controller.rb +59 -0
- data/app/controllers/railwatch/releases_controller.rb +75 -0
- data/app/controllers/railwatch/requests_controller.rb +55 -0
- data/app/controllers/railwatch/saved_views_controller.rb +53 -0
- data/app/controllers/railwatch/scheduled_tasks_controller.rb +45 -0
- data/app/controllers/railwatch/spans_controller.rb +47 -0
- data/app/controllers/railwatch/storage_ops_controller.rb +15 -0
- data/app/controllers/railwatch/tenants_controller.rb +44 -0
- data/app/controllers/railwatch/thresholds_controller.rb +40 -0
- data/app/controllers/railwatch/traces_controller.rb +72 -0
- data/app/controllers/railwatch/transactions_controller.rb +21 -0
- data/app/controllers/railwatch/view_renders_controller.rb +28 -0
- data/app/controllers/railwatch/visits_controller.rb +53 -0
- data/app/helpers/railwatch/assets_helper.rb +52 -0
- data/app/jobs/railwatch/anomaly_scan_job.rb +11 -0
- data/app/jobs/railwatch/application_job.rb +7 -0
- data/app/jobs/railwatch/auto_resolve_issues_job.rb +19 -0
- data/app/jobs/railwatch/check_scheduled_tasks_job.rb +113 -0
- data/app/jobs/railwatch/detect_anomalies_job.rb +171 -0
- data/app/jobs/railwatch/detect_performance_issues_job.rb +85 -0
- data/app/jobs/railwatch/group_exceptions_job.rb +91 -0
- data/app/jobs/railwatch/optimize_telemetry_job.rb +23 -0
- data/app/jobs/railwatch/performance_scan_job.rb +11 -0
- data/app/jobs/railwatch/prune_telemetry_job.rb +110 -0
- data/app/jobs/railwatch/release_health_rollup_job.rb +64 -0
- data/app/jobs/railwatch/rollup_catchup_job.rb +21 -0
- data/app/jobs/railwatch/rollup_job.rb +130 -0
- data/app/jobs/railwatch/scheduled_task_scan_job.rb +11 -0
- data/app/models/concerns/railwatch/detection_snapshotting.rb +20 -0
- data/app/models/railwatch/alert.rb +229 -0
- data/app/models/railwatch/alert_rule.rb +121 -0
- data/app/models/railwatch/anomaly_rule.rb +33 -0
- data/app/models/railwatch/application.rb +44 -0
- data/app/models/railwatch/application_record.rb +33 -0
- data/app/models/railwatch/comment.rb +36 -0
- data/app/models/railwatch/deploy.rb +67 -0
- data/app/models/railwatch/environment.rb +54 -0
- data/app/models/railwatch/execution_presenter.rb +185 -0
- data/app/models/railwatch/filter_query.rb +143 -0
- data/app/models/railwatch/followup_receipt.rb +27 -0
- data/app/models/railwatch/ingest/batch.rb +305 -0
- data/app/models/railwatch/ingest/mapper.rb +575 -0
- data/app/models/railwatch/ingest/payload.rb +96 -0
- data/app/models/railwatch/ingest/rollup_absorber.rb +170 -0
- data/app/models/railwatch/ingest/writer.rb +137 -0
- data/app/models/railwatch/issue.rb +277 -0
- data/app/models/railwatch/issue_activity.rb +23 -0
- data/app/models/railwatch/issue_detection_presenter.rb +309 -0
- data/app/models/railwatch/issue_detection_snapshot.rb +77 -0
- data/app/models/railwatch/maintenance_task.rb +53 -0
- data/app/models/railwatch/saved_view.rb +63 -0
- data/app/models/railwatch/telemetry/aggregations.rb +116 -0
- data/app/models/railwatch/telemetry/attachment.rb +31 -0
- data/app/models/railwatch/telemetry/bounded_gzip.rb +101 -0
- data/app/models/railwatch/telemetry/broadcast.rb +13 -0
- data/app/models/railwatch/telemetry/cache_event.rb +16 -0
- data/app/models/railwatch/telemetry/child.rb +53 -0
- data/app/models/railwatch/telemetry/cursor_page.rb +99 -0
- data/app/models/railwatch/telemetry/deprecation.rb +13 -0
- data/app/models/railwatch/telemetry/enqueued_job.rb +13 -0
- data/app/models/railwatch/telemetry/exception.rb +21 -0
- data/app/models/railwatch/telemetry/execution.rb +71 -0
- data/app/models/railwatch/telemetry/health_sample.rb +72 -0
- data/app/models/railwatch/telemetry/ingest_batch.rb +31 -0
- data/app/models/railwatch/telemetry/llm_call.rb +37 -0
- data/app/models/railwatch/telemetry/log.rb +85 -0
- data/app/models/railwatch/telemetry/mail.rb +13 -0
- data/app/models/railwatch/telemetry/n_plus_one.rb +92 -0
- data/app/models/railwatch/telemetry/notification.rb +13 -0
- data/app/models/railwatch/telemetry/outgoing_request.rb +13 -0
- data/app/models/railwatch/telemetry/person.rb +50 -0
- data/app/models/railwatch/telemetry/process.rb +10 -0
- data/app/models/railwatch/telemetry/profile.rb +37 -0
- data/app/models/railwatch/telemetry/query.rb +50 -0
- data/app/models/railwatch/telemetry/query_shape.rb +45 -0
- data/app/models/railwatch/telemetry/release_health.rb +63 -0
- data/app/models/railwatch/telemetry/rollup.rb +78 -0
- data/app/models/railwatch/telemetry/session.rb +24 -0
- data/app/models/railwatch/telemetry/span.rb +19 -0
- data/app/models/railwatch/telemetry/storage_op.rb +13 -0
- data/app/models/railwatch/telemetry/tenant.rb +221 -0
- data/app/models/railwatch/telemetry/transaction.rb +13 -0
- data/app/models/railwatch/telemetry/view_render.rb +13 -0
- data/app/models/railwatch/telemetry/visit.rb +24 -0
- data/app/models/railwatch/telemetry_record.rb +57 -0
- data/app/models/railwatch/threshold.rb +29 -0
- data/app/models/railwatch/user.rb +47 -0
- data/app/models/railwatch/viewer.rb +13 -0
- data/app/views/layouts/railwatch/dashboard.html.erb +25 -0
- data/config/routes.rb +59 -0
- data/db/railwatch_migrate/20260916000000_create_railwatch_tables.rb +151 -0
- data/db/railwatch_migrate/20260917000000_create_railwatch_maintenance_tasks.rb +18 -0
- data/db/railwatch_migrate/20260917120000_create_railwatch_followup_receipts.rb +20 -0
- data/db/railwatch_migrate/20260918120000_widen_host_user_ids.rb +71 -0
- data/db/railwatch_telemetry_migrate/20260903000001_create_telemetry.rb +481 -0
- data/db/railwatch_telemetry_migrate/20260903000002_rename_tenant_to_app_tenant.rb +14 -0
- data/db/railwatch_telemetry_migrate/20260903000003_add_statement_count_to_transactions.rb +9 -0
- data/db/railwatch_telemetry_migrate/20260903000004_add_role_and_channel.rb +10 -0
- data/db/railwatch_telemetry_migrate/20260903000005_add_locals_to_exceptions.rb +9 -0
- data/db/railwatch_telemetry_migrate/20260903000006_add_spans_health_vitals_and_fts.rb +82 -0
- data/db/railwatch_telemetry_migrate/20260903000007_rename_span_attributes_to_payload.rb +10 -0
- data/db/railwatch_telemetry_migrate/20260903000008_add_profiles_and_attachments.rb +60 -0
- data/db/railwatch_telemetry_migrate/20260903000009_add_truncated_to_attachments.rb +9 -0
- data/db/railwatch_telemetry_migrate/20260903000010_create_sessions_and_release_health.rb +51 -0
- data/db/railwatch_telemetry_migrate/20260903000011_add_fingerprint_to_exceptions.rb +11 -0
- data/db/railwatch_telemetry_migrate/20260904000012_add_failed_to_broadcasts.rb +10 -0
- data/db/railwatch_telemetry_migrate/20260904010000_add_filter_cursor_indexes.rb +37 -0
- data/db/railwatch_telemetry_migrate/20260904020000_add_n_plus_ones_execution_id_index.rb +11 -0
- data/db/railwatch_telemetry_migrate/20260904120000_create_query_shapes.rb +15 -0
- data/db/railwatch_telemetry_migrate/20260906120000_add_backpressure_factor_to_ingest_batches.rb +9 -0
- data/db/railwatch_telemetry_migrate/20260907000000_rename_lantern_version_on_processes.rb +10 -0
- data/db/railwatch_telemetry_migrate/20260913000000_rename_nightrail_version_on_processes.rb +9 -0
- data/db/railwatch_telemetry_migrate/20260914000000_drop_orphan_durable_ingest_tables.rb +18 -0
- data/db/railwatch_telemetry_migrate/20260914010000_drop_orphan_durable_ingest_columns.rb +25 -0
- data/db/railwatch_telemetry_migrate/20260915000000_create_llm_calls.rb +55 -0
- data/db/railwatch_telemetry_migrate/20260915120000_add_detail_to_llm_calls.rb +21 -0
- data/db/railwatch_telemetry_migrate/20260915200000_add_explained_index_to_queries.rb +16 -0
- data/db/railwatch_telemetry_migrate/20260915210000_add_slowest_index_to_queries.rb +18 -0
- data/db/railwatch_telemetry_migrate/20260917010000_add_batch_ledger_to_ingest_batches.rb +16 -0
- data/docs/configuration.md +26 -0
- data/docs/embedded.md +352 -0
- data/docs/getting-started.md +5 -0
- data/lib/generators/railwatch/install/install_generator.rb +222 -1
- data/lib/generators/railwatch/install/templates/{initializer.rb → initializer.rb.tt} +39 -0
- data/lib/generators/railwatch/install/templates/post-deploy +8 -0
- data/lib/puma/plugin/railwatch.rb +170 -0
- data/lib/railwatch/authentication.rb +83 -0
- data/lib/railwatch/configuration.rb +134 -3
- data/lib/railwatch/dashboard_assets.rb +45 -0
- data/lib/railwatch/embedded.rb +54 -0
- data/lib/railwatch/engine.rb +89 -1
- data/lib/railwatch/ingest_request_body_limit.rb +10 -0
- data/lib/railwatch/json_compat.rb +60 -0
- data/lib/railwatch/maintenance.rb +183 -0
- data/lib/railwatch/patches/runner_command.rb +21 -1
- data/lib/railwatch/record.rb +33 -8
- data/lib/railwatch/reporter.rb +16 -0
- data/lib/railwatch/subscribers/process_info.rb +2 -1
- data/lib/railwatch/transport/local.rb +78 -0
- data/lib/railwatch/transport/socket.rb +183 -0
- data/lib/railwatch/version.rb +1 -1
- data/lib/railwatch/writer.rb +370 -0
- data/lib/railwatch.rb +38 -1
- data/lib/tasks/railwatch_tasks.rake +128 -0
- data/public/railwatch/assets/CommitMono-Bold-D6h61ieg.woff2 +0 -0
- data/public/railwatch/assets/CommitMono-Regular-zr8w7Obm.woff2 +0 -0
- data/public/railwatch/assets/Roboto-Black-auA4GeOK.woff2 +0 -0
- data/public/railwatch/assets/Roboto-Bold-CJLnO8j1.woff2 +0 -0
- data/public/railwatch/assets/Roboto-Medium-Cm2bwKpj.woff2 +0 -0
- data/public/railwatch/assets/Roboto-Regular-Chaq1-PV.woff2 +0 -0
- data/public/railwatch/assets/app-layout-yh-sPWgK.js +1 -0
- data/public/railwatch/assets/app-wordmark-nbjkzxwQ.js +1 -0
- data/public/railwatch/assets/appearance-CDuRvQTB.js +1 -0
- data/public/railwatch/assets/application-C_kpBdnf.css +1 -0
- data/public/railwatch/assets/arrow-up-DVtOGdVA.js +1 -0
- data/public/railwatch/assets/auth-layout-TCwPpS1K.js +1 -0
- data/public/railwatch/assets/badge-Daw4Hvr8.js +1 -0
- data/public/railwatch/assets/braces-DhFbHPsz.js +1 -0
- data/public/railwatch/assets/card-BX_3HXcJ.js +1 -0
- data/public/railwatch/assets/chart-B14-N9g7.js +39 -0
- data/public/railwatch/assets/chart-hover-Vy52H4uD.js +1 -0
- data/public/railwatch/assets/chart-panel-Cb4S_ej_.js +1 -0
- data/public/railwatch/assets/checkbox-D3aRSsBj.js +1 -0
- data/public/railwatch/assets/code-KvW8k7Jr.js +1 -0
- data/public/railwatch/assets/copy-DaQMJWoT.js +1 -0
- data/public/railwatch/assets/copy-block-BKFGd11J.js +1 -0
- data/public/railwatch/assets/copy-id-Djks1fXB.js +1 -0
- data/public/railwatch/assets/cursor-load-more-BXV0f1D_.js +1 -0
- data/public/railwatch/assets/data-table-iv1bdF6u.js +1 -0
- data/public/railwatch/assets/edit-BZc_Iawe.js +1 -0
- data/public/railwatch/assets/edit-DMKUF8Zi.js +8 -0
- data/public/railwatch/assets/edit-__9yJlO3.js +1 -0
- data/public/railwatch/assets/empty-state-6j_0AaQQ.js +1 -0
- data/public/railwatch/assets/env-layout-DrT8rO6P.js +1 -0
- data/public/railwatch/assets/execution-path-CzgBUi5e.js +1 -0
- data/public/railwatch/assets/filter-bar-9SU5NrzX.js +1 -0
- data/public/railwatch/assets/flamegraph-qcekju8V.js +2 -0
- data/public/railwatch/assets/format-B9SDkrWj.js +1 -0
- data/public/railwatch/assets/frames-Cyu7KMxZ.js +1 -0
- data/public/railwatch/assets/google-sign-in-button-DsTSfmzY.js +1 -0
- data/public/railwatch/assets/index-1ol1-QWI.js +1 -0
- data/public/railwatch/assets/index-5jI4aFzC.js +1 -0
- data/public/railwatch/assets/index-9KTrVnrc.js +1 -0
- data/public/railwatch/assets/index-B0-8lcTp.js +1 -0
- data/public/railwatch/assets/index-B7jjfNfO.js +1 -0
- data/public/railwatch/assets/index-BBchRy0M.js +1 -0
- data/public/railwatch/assets/index-BRiq3SNR.js +1 -0
- data/public/railwatch/assets/index-BgKj9xhr.js +1 -0
- data/public/railwatch/assets/index-BkTZqqOu.js +1 -0
- data/public/railwatch/assets/index-BprKx8QO.js +1 -0
- data/public/railwatch/assets/index-C3jzvPs3.js +1 -0
- data/public/railwatch/assets/index-Cbs6gGyQ.js +1 -0
- data/public/railwatch/assets/index-CdRZ6AWF.js +1 -0
- data/public/railwatch/assets/index-CeYKnapu.js +1 -0
- data/public/railwatch/assets/index-Cmlwy1-V.js +1 -0
- data/public/railwatch/assets/index-Cwx6058d.js +1 -0
- data/public/railwatch/assets/index-D4CSdbHv.js +1 -0
- data/public/railwatch/assets/index-DEFMSkdG.js +1 -0
- data/public/railwatch/assets/index-DFiHEBSh.js +1 -0
- data/public/railwatch/assets/index-DL4vWdWJ.js +1 -0
- data/public/railwatch/assets/index-DU9F5b5d.js +1 -0
- data/public/railwatch/assets/index-DaXgPcGL.js +1 -0
- data/public/railwatch/assets/index-DbtaU-EE.js +1 -0
- data/public/railwatch/assets/index-DeOe83F4.js +1 -0
- data/public/railwatch/assets/index-DiucHN4B.js +1 -0
- data/public/railwatch/assets/index-DlnR_l9o.js +1 -0
- data/public/railwatch/assets/index-DlumCsWY.js +2 -0
- data/public/railwatch/assets/index-DmRd7aIG.js +1 -0
- data/public/railwatch/assets/index-DxSh2UpM.js +1 -0
- data/public/railwatch/assets/index-MIMGuFNt.js +1 -0
- data/public/railwatch/assets/index-OqI59zPb.js +1 -0
- data/public/railwatch/assets/index-P4rC7IlX.js +1 -0
- data/public/railwatch/assets/index-gpPOcFWq.js +1 -0
- data/public/railwatch/assets/index-oVkururr.js +1 -0
- data/public/railwatch/assets/index-p9puqVge.js +1 -0
- data/public/railwatch/assets/inertia-TViv6kNv.js +97 -0
- data/public/railwatch/assets/input-error-LxImUkxv.js +1 -0
- data/public/railwatch/assets/json-viewer-Ar4cjPDW.js +1 -0
- data/public/railwatch/assets/klass-CJ-J4INB.js +1 -0
- data/public/railwatch/assets/label-COUKWqE_.js +1 -0
- data/public/railwatch/assets/layout-DNSLAkw_.js +1 -0
- data/public/railwatch/assets/live-dot-ChfUtY3p.js +41 -0
- data/public/railwatch/assets/nav-CNnDqPlm.js +1 -0
- data/public/railwatch/assets/new-84S8ZJq9.js +1 -0
- data/public/railwatch/assets/new-Be55nmt9.js +1 -0
- data/public/railwatch/assets/new-Bi_xQiIb.js +1 -0
- data/public/railwatch/assets/new-BvCT8TMg.js +1 -0
- data/public/railwatch/assets/new-D05SajFR.js +1 -0
- data/public/railwatch/assets/new-D4uewYC8.js +1 -0
- data/public/railwatch/assets/onboarding-CYZi5Cqc.js +1 -0
- data/public/railwatch/assets/origin-identity-6q1-CBts.js +1 -0
- data/public/railwatch/assets/percentile-picker-gFZCXtdb.js +1 -0
- data/public/railwatch/assets/relative-time-IOOgl5n2.js +1 -0
- data/public/railwatch/assets/release-health-DC8oc7uw.js +1 -0
- data/public/railwatch/assets/route-Dv6LAWvT.js +1 -0
- data/public/railwatch/assets/segmented-h1VdDTqE.js +1 -0
- data/public/railwatch/assets/select-_AJsUa7X.js +1 -0
- data/public/railwatch/assets/separator-BwwTYtCF.js +1 -0
- data/public/railwatch/assets/series-chart-DaFPefku.js +1 -0
- data/public/railwatch/assets/show-B7NCgkEo.js +1 -0
- data/public/railwatch/assets/show-BKqyKjBK.js +1 -0
- data/public/railwatch/assets/show-BM6X2Mpo.js +1 -0
- data/public/railwatch/assets/show-BNw4tN5q.js +1 -0
- data/public/railwatch/assets/show-BO3bnG5h.js +1 -0
- data/public/railwatch/assets/show-BhrAVAEA.js +1 -0
- data/public/railwatch/assets/show-C4Ltf5i9.js +2 -0
- data/public/railwatch/assets/show-C8sHalnw.js +1 -0
- data/public/railwatch/assets/show-CeTL4B37.js +2 -0
- data/public/railwatch/assets/show-CpfgV1jP.js +1 -0
- data/public/railwatch/assets/show-DACku6AD.js +3 -0
- data/public/railwatch/assets/show-DIOSGcXV.js +6 -0
- data/public/railwatch/assets/show-DQp_1n-B.js +1 -0
- data/public/railwatch/assets/show-DVNz46RI.js +1 -0
- data/public/railwatch/assets/show-DYteoYWW.js +1 -0
- data/public/railwatch/assets/show-DgSIoRvA.js +1 -0
- data/public/railwatch/assets/show-JxFtB4eK.js +2 -0
- data/public/railwatch/assets/sort-header-DpFzXblu.js +1 -0
- data/public/railwatch/assets/source-link-B2183i2-.js +1 -0
- data/public/railwatch/assets/sparkline-cell-C3-5vFkP.js +1 -0
- data/public/railwatch/assets/stat-s4RpOS9w.js +1 -0
- data/public/railwatch/assets/status-badge-8jVV-LA4.js +1 -0
- data/public/railwatch/assets/tenant-path-G-6u9A-o.js +1 -0
- data/public/railwatch/assets/text-link-DfsiaCcP.js +1 -0
- data/public/railwatch/assets/textarea-Dye72uP7.js +1 -0
- data/public/railwatch/assets/timeline-CD7WHnbo.js +1 -0
- data/public/railwatch/assets/transition-B_AW8rMK.js +5 -0
- data/public/railwatch/assets/use-clipboard-ColgLyQ2.js +1 -0
- data/public/railwatch/icon.png +0 -0
- data/public/railwatch/icon.svg +5 -0
- data/public/railwatch/manifest.json +2171 -0
- data/public/railwatch/rails-vite.json +1 -0
- metadata +314 -4
|
@@ -0,0 +1,309 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Railwatch
|
|
4
|
+
# Builds the same discriminated performance/anomaly detail contract for the
|
|
5
|
+
# Inertia page, REST API, and MCP. The durable detector snapshot remains useful
|
|
6
|
+
# after raw data is pruned; live rollups add trend and deploy comparison when
|
|
7
|
+
# they are still available.
|
|
8
|
+
class IssueDetectionPresenter
|
|
9
|
+
TREND_DAYS = 30
|
|
10
|
+
DEPLOY_LIMIT = 20
|
|
11
|
+
EXECUTION_TYPES = IssueDetectionSnapshot::EXECUTION_TYPES
|
|
12
|
+
SCHEMA_VERSION = 1
|
|
13
|
+
NUMERIC_FIELDS = {
|
|
14
|
+
"measurement" => %w[value limit],
|
|
15
|
+
"baseline" => %w[mean stddev sigmas days deviation]
|
|
16
|
+
}.freeze
|
|
17
|
+
RULE_STRING_FIELDS = %w[type target_kind target task_key schedule last_run expected].freeze
|
|
18
|
+
EXECUTION_SOURCES = %w[request job job_attempt scheduled_task command].freeze
|
|
19
|
+
CHILD_TYPES = %w[query outgoing_request].freeze
|
|
20
|
+
TARGET_KIND_FOR = {
|
|
21
|
+
"request" => "requests", "job_attempt" => "jobs", "command" => "commands",
|
|
22
|
+
"query" => "queries", "scheduled_task" => "scheduled_tasks",
|
|
23
|
+
"outgoing_request" => "outgoing_requests"
|
|
24
|
+
}.freeze
|
|
25
|
+
|
|
26
|
+
def initialize(issue)
|
|
27
|
+
@issue = issue
|
|
28
|
+
@details = issue.sample["detection"]
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
def as_json
|
|
32
|
+
return unless @issue.kind.in?(%w[performance anomaly])
|
|
33
|
+
return unavailable_json unless @details.is_a?(Hash)
|
|
34
|
+
|
|
35
|
+
validate_details!
|
|
36
|
+
live = @issue.environment.with_telemetry { live_details }
|
|
37
|
+
@details.deep_symbolize_keys.merge(
|
|
38
|
+
available: true,
|
|
39
|
+
count_label: "Breached evaluation windows",
|
|
40
|
+
breached_windows: @issue.occurrences,
|
|
41
|
+
event_count: live[:event_count],
|
|
42
|
+
trend: live[:trend],
|
|
43
|
+
deploy_comparison: live[:deploy_comparison],
|
|
44
|
+
representative_references: representative_references,
|
|
45
|
+
representative_records: representatives
|
|
46
|
+
)
|
|
47
|
+
rescue KeyError, ArgumentError, TypeError
|
|
48
|
+
unavailable_json
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
private
|
|
52
|
+
|
|
53
|
+
def unavailable_json
|
|
54
|
+
{
|
|
55
|
+
kind: @issue.kind, available: false,
|
|
56
|
+
count_label: "Breached evaluation windows", breached_windows: @issue.occurrences,
|
|
57
|
+
message: "Typed detector context was not captured for this historical issue."
|
|
58
|
+
}
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
def validate_details!
|
|
62
|
+
raise ArgumentError unless @details.fetch("schema_version") == SCHEMA_VERSION
|
|
63
|
+
raise ArgumentError unless @details.fetch("kind") == @issue.kind
|
|
64
|
+
telemetry_type = @details.fetch("telemetry_type")
|
|
65
|
+
raise ArgumentError unless telemetry_type.in?(EXECUTION_TYPES + %w[query outgoing_request])
|
|
66
|
+
target_kind = TARGET_KIND_FOR.fetch(telemetry_type)
|
|
67
|
+
allowed_target_kinds = @issue.kind == "anomaly" ? AnomalyRule::TARGET_KINDS : Threshold::TARGET_KINDS
|
|
68
|
+
raise ArgumentError unless target_kind.in?(allowed_target_kinds)
|
|
69
|
+
%w[telemetry_group_hash target metric unit].each do |key|
|
|
70
|
+
raise ArgumentError unless @details.fetch(key).is_a?(String) && @details.fetch(key).present?
|
|
71
|
+
end
|
|
72
|
+
metric = @details.fetch("metric")
|
|
73
|
+
allowed_metrics = @issue.kind == "anomaly" ? AnomalyRule::METRICS : Threshold::METRICS
|
|
74
|
+
raise ArgumentError unless metric.in?(allowed_metrics)
|
|
75
|
+
raise ArgumentError if metric == "missed" && @details.fetch("telemetry_type") != "scheduled_task"
|
|
76
|
+
expected_unit = IssueDetectionSnapshot::UNITS[metric]
|
|
77
|
+
raise ArgumentError unless expected_unit && @details.fetch("unit") == expected_unit
|
|
78
|
+
validate_window!
|
|
79
|
+
NUMERIC_FIELDS.each { |key, fields| validate_numeric_object!(key, fields) }
|
|
80
|
+
validate_rule!
|
|
81
|
+
validate_detector_shape!
|
|
82
|
+
validate_representatives!
|
|
83
|
+
end
|
|
84
|
+
|
|
85
|
+
def validate_window!
|
|
86
|
+
window = @details.fetch("window")
|
|
87
|
+
raise ArgumentError unless window.is_a?(Hash)
|
|
88
|
+
from = Time.iso8601(window.fetch("from"))
|
|
89
|
+
to = Time.iso8601(window.fetch("to"))
|
|
90
|
+
minutes = window.fetch("minutes")
|
|
91
|
+
raise ArgumentError unless finite_number?(minutes) && minutes.positive? && to > from
|
|
92
|
+
raise ArgumentError unless (minutes - ((to - from) / 60.0).round(2)).abs <= 0.01
|
|
93
|
+
return if @details.fetch("metric") == "missed"
|
|
94
|
+
|
|
95
|
+
raise ArgumentError unless minutes == minutes.to_i
|
|
96
|
+
maximum = @issue.kind == "anomaly" ? AnomalyRule::MAX_WINDOW_MINUTES : Threshold::MAX_WINDOW_MINUTES
|
|
97
|
+
raise ArgumentError if minutes > maximum
|
|
98
|
+
end
|
|
99
|
+
|
|
100
|
+
def validate_numeric_object!(key, fields)
|
|
101
|
+
value = @details[key]
|
|
102
|
+
return if value.nil?
|
|
103
|
+
raise ArgumentError unless value.is_a?(Hash)
|
|
104
|
+
fields.each { |field| raise ArgumentError if value.key?(field) && !finite_number?(value[field]) }
|
|
105
|
+
if key == "measurement"
|
|
106
|
+
%w[value limit].each { |field| raise ArgumentError if value.key?(field) && value[field].negative? }
|
|
107
|
+
raise ArgumentError if value.key?("limit") && !value["limit"].positive?
|
|
108
|
+
else
|
|
109
|
+
%w[mean stddev].each { |field| raise ArgumentError if value.key?(field) && value[field].negative? }
|
|
110
|
+
raise ArgumentError if value.key?("days") && (!value["days"].is_a?(Integer) || !value["days"].positive?)
|
|
111
|
+
raise ArgumentError if value.key?("deviation") && !value["deviation"].positive?
|
|
112
|
+
end
|
|
113
|
+
end
|
|
114
|
+
|
|
115
|
+
def validate_rule!
|
|
116
|
+
rule = @details["rule"]
|
|
117
|
+
if rule.nil?
|
|
118
|
+
raise ArgumentError if @details.fetch("metric") == "missed"
|
|
119
|
+
return
|
|
120
|
+
end
|
|
121
|
+
raise ArgumentError unless rule.is_a?(Hash)
|
|
122
|
+
RULE_STRING_FIELDS.each do |field|
|
|
123
|
+
raise ArgumentError if rule.key?(field) && (!rule[field].is_a?(String) || rule[field].blank?)
|
|
124
|
+
end
|
|
125
|
+
raise ArgumentError if rule.key?("id") && (!rule["id"].is_a?(Integer) || !rule["id"].positive?)
|
|
126
|
+
%w[last_run expected].each { |field| Time.iso8601(rule[field]) if rule.key?(field) }
|
|
127
|
+
return unless @details.fetch("metric") == "missed"
|
|
128
|
+
|
|
129
|
+
raise ArgumentError unless rule["type"] == "schedule"
|
|
130
|
+
%w[task_key schedule last_run expected].each { |field| raise ArgumentError unless rule[field].is_a?(String) && rule[field].present? }
|
|
131
|
+
end
|
|
132
|
+
|
|
133
|
+
def validate_detector_shape!
|
|
134
|
+
if @issue.kind == "anomaly"
|
|
135
|
+
require_numeric_fields!("measurement", %w[value])
|
|
136
|
+
baseline = require_numeric_fields!("baseline", %w[mean stddev sigmas days deviation])
|
|
137
|
+
raise ArgumentError unless baseline.fetch("sigmas").positive?
|
|
138
|
+
raise ArgumentError unless baseline.fetch("days").between?(
|
|
139
|
+
AnomalyRule::MIN_BASELINE_DAYS, AnomalyRule::MAX_BASELINE_DAYS
|
|
140
|
+
)
|
|
141
|
+
raise ArgumentError if baseline.fetch("deviation") > AnomalyRule::MAX_DEVIATION
|
|
142
|
+
require_detector_rule!("anomaly")
|
|
143
|
+
elsif @details.fetch("metric") != "missed"
|
|
144
|
+
measurement = require_numeric_fields!("measurement", %w[value limit])
|
|
145
|
+
raise ArgumentError if measurement.fetch("limit") > Threshold::MAX_LIMIT
|
|
146
|
+
raise ArgumentError if @details.fetch("metric").end_with?("rate") && measurement.fetch("limit") > 100
|
|
147
|
+
require_detector_rule!("threshold")
|
|
148
|
+
end
|
|
149
|
+
end
|
|
150
|
+
|
|
151
|
+
def require_numeric_fields!(key, fields)
|
|
152
|
+
object = @details.fetch(key)
|
|
153
|
+
raise ArgumentError unless object.is_a?(Hash)
|
|
154
|
+
fields.each { |field| raise ArgumentError unless finite_number?(object.fetch(field)) }
|
|
155
|
+
object
|
|
156
|
+
end
|
|
157
|
+
|
|
158
|
+
def require_detector_rule!(type)
|
|
159
|
+
rule = @details.fetch("rule")
|
|
160
|
+
raise ArgumentError unless rule.is_a?(Hash) && rule.fetch("type") == type
|
|
161
|
+
raise ArgumentError unless rule.fetch("id").is_a?(Integer) && rule.fetch("id").positive?
|
|
162
|
+
raise ArgumentError unless rule.fetch("target").is_a?(String) && rule.fetch("target").present?
|
|
163
|
+
expected_target_kind = TARGET_KIND_FOR.fetch(@details.fetch("telemetry_type"))
|
|
164
|
+
raise ArgumentError unless rule.fetch("target_kind") == expected_target_kind
|
|
165
|
+
end
|
|
166
|
+
|
|
167
|
+
def validate_representatives!
|
|
168
|
+
references = @details.fetch("representative_records", [])
|
|
169
|
+
raise ArgumentError unless references.is_a?(Array) && references.size <= IssueDetectionSnapshot::REPRESENTATIVE_LIMIT
|
|
170
|
+
|
|
171
|
+
references.each do |reference|
|
|
172
|
+
raise ArgumentError unless reference.is_a?(Hash)
|
|
173
|
+
raise ArgumentError unless reference["record_type"] == @details.fetch("telemetry_type")
|
|
174
|
+
raise ArgumentError unless reference["record_id"].is_a?(Integer) && reference["record_id"].positive?
|
|
175
|
+
raise ArgumentError unless reference["group_hash"] == @details.fetch("telemetry_group_hash")
|
|
176
|
+
execution_id = reference["execution_id"]
|
|
177
|
+
raise ArgumentError unless execution_id.nil? || (execution_id.is_a?(String) && execution_id.present?)
|
|
178
|
+
raise ArgumentError if reference["record_type"] != "query" && execution_id.blank?
|
|
179
|
+
source = reference["execution_source"]
|
|
180
|
+
if CHILD_TYPES.include?(reference["record_type"])
|
|
181
|
+
raise ArgumentError unless source.is_a?(String) && source.present? && source.in?(EXECUTION_SOURCES)
|
|
182
|
+
else
|
|
183
|
+
raise ArgumentError unless source == reference["record_type"]
|
|
184
|
+
end
|
|
185
|
+
end
|
|
186
|
+
end
|
|
187
|
+
|
|
188
|
+
def finite_number?(value)
|
|
189
|
+
value.is_a?(Numeric) && (!value.respond_to?(:finite?) || value.finite?)
|
|
190
|
+
end
|
|
191
|
+
|
|
192
|
+
def live_details
|
|
193
|
+
type = @details.fetch("telemetry_type")
|
|
194
|
+
group_hash = @details.fetch("telemetry_group_hash")
|
|
195
|
+
window = @details.fetch("window")
|
|
196
|
+
from = Time.iso8601(window.fetch("from"))
|
|
197
|
+
to = Time.iso8601(window.fetch("to"))
|
|
198
|
+
rollups = Telemetry::Rollup.for_type(type).where(group_hash: group_hash)
|
|
199
|
+
window_summary = Telemetry::Rollup.summarize(rollups.between(from, to))
|
|
200
|
+
now = Time.current
|
|
201
|
+
# Rollup.between includes the whole lower-bound hour, so the denominator
|
|
202
|
+
# for count rates must begin at that same boundary.
|
|
203
|
+
trend_from = (now - TREND_DAYS.days).beginning_of_hour
|
|
204
|
+
trend_rows = rollups.between(trend_from, now).order(:bucket).to_a.group_by { |row| row.bucket.to_date }
|
|
205
|
+
{
|
|
206
|
+
event_count: window_summary[:count],
|
|
207
|
+
trend: trend_rows.map { |day, rows| trend_point(day, rows, trend_from, now) },
|
|
208
|
+
deploy_comparison: deploy_comparison(type, group_hash)
|
|
209
|
+
}
|
|
210
|
+
end
|
|
211
|
+
|
|
212
|
+
def trend_point(day, rows, trend_from, now)
|
|
213
|
+
summary = Telemetry::Rollup.summarize(rows)
|
|
214
|
+
day_start = day.in_time_zone.beginning_of_day
|
|
215
|
+
minutes = ([ day_start + 1.day, now ].min - [ day_start, trend_from ].max) / 60.0
|
|
216
|
+
{ day: day.iso8601, value: metric_value(@details.fetch("metric"), summary, minutes: minutes), count: summary[:count] }
|
|
217
|
+
end
|
|
218
|
+
|
|
219
|
+
def metric_value(metric, summary, minutes: nil)
|
|
220
|
+
case metric
|
|
221
|
+
when "p95" then milliseconds(summary[:p95])
|
|
222
|
+
when "max" then milliseconds(summary[:max])
|
|
223
|
+
when "avg" then milliseconds(summary[:avg])
|
|
224
|
+
when "error_rate", "failure_rate"
|
|
225
|
+
summary[:count].zero? ? nil : (summary[:errors] * 100.0 / summary[:count]).round(2)
|
|
226
|
+
when "count" then minutes&.positive? ? (summary[:count] / minutes).round(2) : nil
|
|
227
|
+
end
|
|
228
|
+
end
|
|
229
|
+
|
|
230
|
+
def deploy_comparison(type, group_hash)
|
|
231
|
+
scope = record_scope(type, group_hash).where(occurred_at: TREND_DAYS.days.ago..).where.not(deploy: nil)
|
|
232
|
+
scope.group(:deploy).pluck(:deploy, Arel.sql("COUNT(*)"), Arel.sql("AVG(duration)"), Arel.sql("MAX(duration)"))
|
|
233
|
+
.map { |deploy, count, avg, max| { deploy: deploy, events: count, avg_ms: milliseconds(avg), max_ms: milliseconds(max) } }
|
|
234
|
+
.sort_by { |row| -row[:events] }.first(DEPLOY_LIMIT)
|
|
235
|
+
end
|
|
236
|
+
|
|
237
|
+
def record_scope(type, group_hash)
|
|
238
|
+
if type.in?(EXECUTION_TYPES)
|
|
239
|
+
Telemetry::Execution.of_kind(type).where(group_hash: group_hash)
|
|
240
|
+
elsif type == "query"
|
|
241
|
+
Telemetry::Query.where(group_hash: group_hash).with_sql
|
|
242
|
+
elsif type == "outgoing_request"
|
|
243
|
+
Telemetry::OutgoingRequest.where(group_hash: group_hash)
|
|
244
|
+
else
|
|
245
|
+
raise ArgumentError, "unsupported detection telemetry type: #{type}"
|
|
246
|
+
end
|
|
247
|
+
end
|
|
248
|
+
|
|
249
|
+
def representatives
|
|
250
|
+
references = Array(@details["representative_records"]).select { |record| record.is_a?(Hash) }
|
|
251
|
+
return [] if references.empty?
|
|
252
|
+
|
|
253
|
+
type = @details.fetch("telemetry_type")
|
|
254
|
+
group_hash = @details.fetch("telemetry_group_hash")
|
|
255
|
+
records = @issue.environment.with_telemetry do
|
|
256
|
+
record_scope(type, group_hash).where(id: references.filter_map { |record| record["record_id"] }).index_by(&:id)
|
|
257
|
+
end
|
|
258
|
+
references.filter_map do |reference|
|
|
259
|
+
record = records[reference["record_id"]]
|
|
260
|
+
record_json(record, type)&.merge(drilldown: drilldown(reference))
|
|
261
|
+
end
|
|
262
|
+
end
|
|
263
|
+
|
|
264
|
+
def representative_references
|
|
265
|
+
Array(@details["representative_records"]).filter_map do |reference|
|
|
266
|
+
next unless reference.is_a?(Hash)
|
|
267
|
+
reference.slice("record_type", "record_id", "group_hash", "execution_id", "execution_source").deep_symbolize_keys
|
|
268
|
+
.merge(drilldown: drilldown(reference))
|
|
269
|
+
end
|
|
270
|
+
end
|
|
271
|
+
|
|
272
|
+
def record_json(record, type)
|
|
273
|
+
return unless record
|
|
274
|
+
|
|
275
|
+
common = {
|
|
276
|
+
record_type: type, record_id: record.id, group_hash: record.group_hash,
|
|
277
|
+
execution_id: record.execution_id, execution_source: record.execution_source,
|
|
278
|
+
execution_preview: record.execution_preview, occurred_at: record.occurred_at,
|
|
279
|
+
duration_ms: milliseconds(record.duration), deploy: record.deploy,
|
|
280
|
+
user_ref: record.user_ref, tenant: record.app_tenant
|
|
281
|
+
}
|
|
282
|
+
case record
|
|
283
|
+
when Telemetry::Execution
|
|
284
|
+
common.merge(name: record.name, status: record.status, outcome: record.outcome)
|
|
285
|
+
when Telemetry::Query
|
|
286
|
+
common.merge(name: record.sql.to_s.first(300), source: record.source,
|
|
287
|
+
connection: record.connection, role: record.role)
|
|
288
|
+
when Telemetry::OutgoingRequest
|
|
289
|
+
common.merge(name: "#{record.method} #{record.host}", status: record.status_code,
|
|
290
|
+
source: record.source)
|
|
291
|
+
end
|
|
292
|
+
end
|
|
293
|
+
|
|
294
|
+
# Identifiers only: app/frontend/lib/execution-path.ts turns these into
|
|
295
|
+
# links, and it already accepts both the "job" child wire name and the
|
|
296
|
+
# "job_attempt" parent kind.
|
|
297
|
+
def drilldown(record)
|
|
298
|
+
if record["record_type"] == "query"
|
|
299
|
+
{ kind: "query", group_hash: record["group_hash"] }
|
|
300
|
+
elsif record["execution_id"].present?
|
|
301
|
+
{ kind: record["execution_source"].presence || record["record_type"], execution_id: record["execution_id"] }
|
|
302
|
+
end
|
|
303
|
+
end
|
|
304
|
+
|
|
305
|
+
def milliseconds(value)
|
|
306
|
+
value.nil? ? nil : (value.to_f / 1000.0).round(3)
|
|
307
|
+
end
|
|
308
|
+
end
|
|
309
|
+
end
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Railwatch
|
|
4
|
+
# Durable, typed context for synthetic performance/anomaly issues. Detector
|
|
5
|
+
# group hashes identify the rule, not the telemetry rows behind the breach;
|
|
6
|
+
# this snapshot preserves that original identity and a bounded set of
|
|
7
|
+
# drill-down identifiers even after raw telemetry is pruned. Record payloads
|
|
8
|
+
# remain subject to the telemetry retention policy and are never copied into
|
|
9
|
+
# the primary database.
|
|
10
|
+
class IssueDetectionSnapshot
|
|
11
|
+
REPRESENTATIVE_LIMIT = 10
|
|
12
|
+
EXECUTION_TYPES = %w[request job_attempt scheduled_task command].freeze
|
|
13
|
+
UNITS = {
|
|
14
|
+
"p95" => "milliseconds", "max" => "milliseconds", "avg" => "milliseconds",
|
|
15
|
+
"error_rate" => "percent", "failure_rate" => "percent", "count" => "events_per_minute",
|
|
16
|
+
"missed" => "schedule"
|
|
17
|
+
}.freeze
|
|
18
|
+
|
|
19
|
+
def self.capture(environment:, kind:, telemetry_type:, telemetry_group_hash:, target:, metric:, from:, to:,
|
|
20
|
+
value: nil, limit: nil, baseline: nil, rule: {}, representative_from: nil)
|
|
21
|
+
new(environment, telemetry_type, telemetry_group_hash).capture(
|
|
22
|
+
kind: kind, target: target, metric: metric, from: from, to: to,
|
|
23
|
+
value: value, limit: limit, baseline: baseline, rule: rule, representative_from: representative_from
|
|
24
|
+
)
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
def initialize(environment, telemetry_type, telemetry_group_hash)
|
|
28
|
+
@environment = environment
|
|
29
|
+
@telemetry_type = telemetry_type
|
|
30
|
+
@telemetry_group_hash = telemetry_group_hash
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
def capture(kind:, target:, metric:, from:, to:, value:, limit:, baseline:, rule:, representative_from:)
|
|
34
|
+
representatives = @environment.with_telemetry do
|
|
35
|
+
representative_scope(representative_from || from, to).order(duration: :desc).limit(REPRESENTATIVE_LIMIT)
|
|
36
|
+
.map { |record| record_json(record) }
|
|
37
|
+
end
|
|
38
|
+
{
|
|
39
|
+
schema_version: 1, kind: kind, telemetry_type: @telemetry_type,
|
|
40
|
+
telemetry_group_hash: @telemetry_group_hash, target: target, metric: metric,
|
|
41
|
+
unit: UNITS.fetch(metric, "value"),
|
|
42
|
+
window: { from: from.iso8601(6), to: to.iso8601(6), minutes: ((to - from) / 60.0).round(2) },
|
|
43
|
+
measurement: { value: value, limit: limit }.compact,
|
|
44
|
+
baseline: baseline&.compact,
|
|
45
|
+
rule: rule.compact,
|
|
46
|
+
representative_records: representatives
|
|
47
|
+
}.compact
|
|
48
|
+
end
|
|
49
|
+
|
|
50
|
+
private
|
|
51
|
+
|
|
52
|
+
def representative_scope(from, to)
|
|
53
|
+
model = record_model
|
|
54
|
+
# SQLite compares its stored six-digit timestamp strings lexically. A
|
|
55
|
+
# frozen whole-second upper bound serializes without the fraction and
|
|
56
|
+
# would otherwise exclude a record at that exact instant.
|
|
57
|
+
scope = model.where(group_hash: @telemetry_group_hash, occurred_at: from...(to + 1.second))
|
|
58
|
+
EXECUTION_TYPES.include?(@telemetry_type) ? scope.of_kind(@telemetry_type) : scope
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
def record_model
|
|
62
|
+
return Telemetry::Execution if EXECUTION_TYPES.include?(@telemetry_type)
|
|
63
|
+
return Telemetry::Query if @telemetry_type == "query"
|
|
64
|
+
return Telemetry::OutgoingRequest if @telemetry_type == "outgoing_request"
|
|
65
|
+
|
|
66
|
+
raise ArgumentError, "unsupported detection telemetry type: #{@telemetry_type}"
|
|
67
|
+
end
|
|
68
|
+
|
|
69
|
+
def record_json(record)
|
|
70
|
+
{
|
|
71
|
+
record_type: @telemetry_type, record_id: record.id, group_hash: record.group_hash,
|
|
72
|
+
execution_id: record.execution_id,
|
|
73
|
+
execution_source: record.is_a?(Telemetry::Execution) ? record.kind : record.execution_source
|
|
74
|
+
}
|
|
75
|
+
end
|
|
76
|
+
end
|
|
77
|
+
end
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Railwatch
|
|
4
|
+
# One row per maintenance task (Railwatch::Maintenance::TASKS): when it last
|
|
5
|
+
# ran and who holds it now. Every process on the host runs the maintenance
|
|
6
|
+
# clock; this is how exactly one of them runs each task. The claim is a
|
|
7
|
+
# single conditional UPDATE, so SQLite's write lock is the arbiter and there
|
|
8
|
+
# is no window between checking and taking the lease.
|
|
9
|
+
class MaintenanceTask < ApplicationRecord
|
|
10
|
+
self.table_name = "railwatch_maintenance_tasks"
|
|
11
|
+
|
|
12
|
+
# The token for a claim this process won, or nil: the task is due
|
|
13
|
+
# (last_run_at older than `every`, or never) and nobody holds a live
|
|
14
|
+
# lease on it. A lease left by a process that died expires on its own.
|
|
15
|
+
# The token is per claim, not per process, so a claim that outlived its
|
|
16
|
+
# lease cannot later release the lease the next claimant holds.
|
|
17
|
+
def self.claim(name, every:, lease:, owner:, now: Time.current)
|
|
18
|
+
ensure_row(name)
|
|
19
|
+
token = "#{owner}:#{SecureRandom.hex(8)}"
|
|
20
|
+
won = where(name: name)
|
|
21
|
+
.where("lease_expires_at IS NULL OR lease_expires_at < ?", now)
|
|
22
|
+
.where("last_run_at IS NULL OR last_run_at <= ?", now - every)
|
|
23
|
+
.update_all(lease_owner: token, lease_expires_at: now + lease, updated_at: now) == 1
|
|
24
|
+
won ? token : nil
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
# Releases only the lease this token holds. A task that succeeded records
|
|
28
|
+
# the run so the interval starts again; one that failed records nothing,
|
|
29
|
+
# so it is eligible on the next tick rather than a full interval later.
|
|
30
|
+
def self.release(name, token:, ran_at:, succeeded:)
|
|
31
|
+
changes = { lease_owner: nil, lease_expires_at: nil, updated_at: ran_at }
|
|
32
|
+
changes[:last_run_at] = ran_at if succeeded
|
|
33
|
+
where(name: name, lease_owner: token).update_all(changes)
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
# The newest tick across every task: what the doctor reports as "last
|
|
37
|
+
# maintenance", and the signal that the clock is alive somewhere.
|
|
38
|
+
def self.last_tick_at
|
|
39
|
+
maximum(:last_run_at)
|
|
40
|
+
end
|
|
41
|
+
|
|
42
|
+
# Two processes racing to create the same row: one loses on the unique
|
|
43
|
+
# index, and the row it wanted now exists.
|
|
44
|
+
def self.ensure_row(name)
|
|
45
|
+
return if exists?(name: name)
|
|
46
|
+
|
|
47
|
+
create!(name: name)
|
|
48
|
+
rescue ActiveRecord::RecordNotUnique
|
|
49
|
+
nil
|
|
50
|
+
end
|
|
51
|
+
private_class_method :ensure_row
|
|
52
|
+
end
|
|
53
|
+
end
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Railwatch
|
|
4
|
+
# A named, shareable filter on one environment page: "5xx checkout requests",
|
|
5
|
+
# "acme tenant logs". Pinned views appear in the sidebar under the page.
|
|
6
|
+
class SavedView < ApplicationRecord
|
|
7
|
+
self.table_name = "railwatch_saved_views"
|
|
8
|
+
# page => the index route helper the view links back to. The link is built
|
|
9
|
+
# server-side so the sidebar, the menu, and a pasted URL all agree.
|
|
10
|
+
PAGE_PATHS = {
|
|
11
|
+
"requests" => :application_environment_requests_path,
|
|
12
|
+
"jobs" => :application_environment_jobs_path,
|
|
13
|
+
"scheduled_tasks" => :application_environment_scheduled_tasks_path,
|
|
14
|
+
"commands" => :application_environment_commands_path,
|
|
15
|
+
"exceptions" => :application_environment_exceptions_path,
|
|
16
|
+
"queries" => :application_environment_queries_path,
|
|
17
|
+
"spans" => :application_environment_spans_path,
|
|
18
|
+
"logs" => :application_environment_logs_path,
|
|
19
|
+
"visits" => :application_environment_visits_path,
|
|
20
|
+
"people" => :application_environment_people_path,
|
|
21
|
+
"tenants" => :application_environment_tenants_path,
|
|
22
|
+
"llm_calls" => :application_environment_llm_calls_path,
|
|
23
|
+
"outgoing_requests" => :application_environment_outgoing_requests_path,
|
|
24
|
+
"cache_events" => :application_environment_cache_events_path,
|
|
25
|
+
"mails" => :application_environment_mails_path,
|
|
26
|
+
"notifications" => :application_environment_notifications_path
|
|
27
|
+
}.freeze
|
|
28
|
+
PAGES = PAGE_PATHS.keys.freeze
|
|
29
|
+
|
|
30
|
+
def environment = Environment.current
|
|
31
|
+
def user
|
|
32
|
+
User.find_by(id: viewer_id)
|
|
33
|
+
end
|
|
34
|
+
|
|
35
|
+
def user=(u)
|
|
36
|
+
self.viewer_id = u&.id&.to_s
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
validates :name, presence: true, length: { maximum: 80 }
|
|
40
|
+
validates :page, inclusion: { in: PAGES }
|
|
41
|
+
|
|
42
|
+
scope :pinned, -> { where(pinned: true) }
|
|
43
|
+
scope :visible_to, ->(user) { where(shared: true).or(where(viewer_id: user.id.to_s)) }
|
|
44
|
+
|
|
45
|
+
# The props every environment page shares, ready for the sidebar and the
|
|
46
|
+
# Views menu: one entry per view this user is allowed to see.
|
|
47
|
+
def self.props_for(environment, user)
|
|
48
|
+
where(environment_id: environment.id).visible_to(user).order(:name).map { |view| view.props(environment, user) }
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
def props(environment, user)
|
|
52
|
+
{ id: id, name: name, page: page, query: query, window: window, params: params, pinned: pinned, shared: shared,
|
|
53
|
+
mine: viewer_id == user&.id, url: path_for(environment) }
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
# This page's index path with the view's window, query, and extra params
|
|
57
|
+
# (sort/dir) applied -- the shareable URL.
|
|
58
|
+
def path_for(environment)
|
|
59
|
+
Railwatch::Engine.routes.url_helpers.public_send(PAGE_PATHS.fetch(page), environment.application_id, environment.id,
|
|
60
|
+
{ window: window.presence, q: query.presence, **params.to_h.symbolize_keys }.compact)
|
|
61
|
+
end
|
|
62
|
+
end
|
|
63
|
+
end
|
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Railwatch
|
|
4
|
+
module Telemetry
|
|
5
|
+
# Pure per-environment rollup aggregation: time-bucketed series, grouped
|
|
6
|
+
# tables, and window-over-window summaries. Takes explicit from/to rather
|
|
7
|
+
# than reading a controller's window params, so it's the same math for the
|
|
8
|
+
# web dashboard (EnvironmentScoped) and the public API (Api::V1).
|
|
9
|
+
class Aggregations
|
|
10
|
+
GROUPED_SORT_FIELDS = %i[count p50 p95 p99 errors avg max].freeze
|
|
11
|
+
|
|
12
|
+
# Time-bucketed series from rollups for charts: [{t, count, errors, p50, p95, p99}]
|
|
13
|
+
def self.series(environment, record_type, from:, to:, group_hash: nil, name: nil)
|
|
14
|
+
environment.with_telemetry do
|
|
15
|
+
scope = Telemetry::Rollup.for_type(record_type).between(from, to)
|
|
16
|
+
scope = scope.where(group_hash: group_hash) if group_hash
|
|
17
|
+
scope = scope.where(name: name) if name
|
|
18
|
+
scope.group(:bucket).order(:bucket)
|
|
19
|
+
.pluck(:bucket, Arel.sql("SUM(count)"), Arel.sql("SUM(error_count)"), Arel.sql("SUM(client_error_count)"), Arel.sql("SUM(duration_sum)"), Arel.sql("MAX(p50)"), Arel.sql("MAX(p95)"), Arel.sql("MAX(p99)"))
|
|
20
|
+
.map { |b, c, e, ce, ds, p50, p95, p99| { t: b, count: c, errors: e, client_errors: ce, avg: c.zero? ? 0 : ds / c / 1000.0, p50: p50 / 1000.0, p95: p95 / 1000.0, p99: p99 / 1000.0 } }
|
|
21
|
+
end
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
# Per-group table rows for a record type in the window: count/error totals,
|
|
25
|
+
# merged percentiles (via Telemetry::Rollup.summarize), and a sparkline.
|
|
26
|
+
def self.grouped(environment, record_type, from:, to:, limit: 100, order: nil, dir: nil)
|
|
27
|
+
order = GROUPED_SORT_FIELDS.include?(order.to_s.to_sym) ? order.to_s.to_sym : :count
|
|
28
|
+
sign = dir.to_s == "asc" ? 1 : -1
|
|
29
|
+
# Same minute cache as summary_with_delta, for the same reason: the
|
|
30
|
+
# 200 busiest query groups on the rebulk environment span 1,500 rollup
|
|
31
|
+
# rows and 55,000 centroids, and merging those digests is 480ms that
|
|
32
|
+
# only changes when RollupJob writes.
|
|
33
|
+
key = [ "aggregations", "grouped", environment.id, record_type, limit, order, sign, from.to_i / 60, to.to_i / 60 ]
|
|
34
|
+
cached(key) { grouped_uncached(environment, record_type, from: from, to: to, limit: limit, order: order, sign: sign) }
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
def self.grouped_uncached(environment, record_type, from:, to:, limit:, order:, sign:)
|
|
38
|
+
environment.with_telemetry do
|
|
39
|
+
rows = Telemetry::Rollup.for_type(record_type).between(from, to).to_a
|
|
40
|
+
groups = rows.group_by(&:group_hash)
|
|
41
|
+
# Merging t-digests is the expensive part (the queries page has 2,000
|
|
42
|
+
# groups over 8,500 rows: 500ms of digest merges for a table that
|
|
43
|
+
# shows 200). Sort on what the columns already hold -- counts, sums
|
|
44
|
+
# and maxima need no digest -- and only merge percentiles for the
|
|
45
|
+
# groups that make the cut. A percentile sort falls back to the full
|
|
46
|
+
# merge, since the ranking itself needs the digest.
|
|
47
|
+
cheap = %i[count errors avg max].include?(order)
|
|
48
|
+
ranked = groups.map { |group_hash, group_rows| [ group_hash, group_rows, cheap ? cheap_summary(group_rows) : Telemetry::Rollup.summarize(group_rows) ] }
|
|
49
|
+
ranked.sort_by! { |_, _, summary| summary[order] * sign }
|
|
50
|
+
ranked.first(limit).map do |group_hash, group_rows, summary|
|
|
51
|
+
summary = Telemetry::Rollup.summarize(group_rows) if cheap
|
|
52
|
+
{
|
|
53
|
+
group_hash: group_hash, name: group_rows.max_by(&:bucket).name,
|
|
54
|
+
count: summary[:count], errors: summary[:errors], client_errors: summary[:client_errors],
|
|
55
|
+
avg: (summary[:avg] / 1000.0).round(2), p50: (summary[:p50] / 1000.0).round(2),
|
|
56
|
+
p95: (summary[:p95] / 1000.0).round(2), p99: (summary[:p99] / 1000.0).round(2),
|
|
57
|
+
max: (summary[:max] / 1000.0).round(2), sparkline: sparkline(group_rows, from, to)
|
|
58
|
+
}
|
|
59
|
+
end
|
|
60
|
+
end
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
# The minute cache exists because the platform's rollups only change
|
|
64
|
+
# when RollupJob writes. Embedded, every batch updates them and one
|
|
65
|
+
# person is looking, so the cache would only make the page a minute
|
|
66
|
+
# stale for nothing.
|
|
67
|
+
def self.cached(key, &block)
|
|
68
|
+
return yield if Railwatch.config.local?
|
|
69
|
+
|
|
70
|
+
Rails.cache.fetch(key, expires_in: 1.minute, &block)
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
# The digest-free half of Rollup.summarize: everything the sort keys
|
|
74
|
+
# that are not percentiles need.
|
|
75
|
+
def self.cheap_summary(rows)
|
|
76
|
+
count = rows.sum(&:count)
|
|
77
|
+
{ count: count, errors: rows.sum(&:error_count),
|
|
78
|
+
avg: count.zero? ? 0 : (rows.sum(&:duration_sum) / count), max: rows.map(&:duration_max).max }
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
# { current: Summary, previous: Summary } so pages/endpoints can show deltas.
|
|
82
|
+
# Cached for a minute per window. The two totals merge every rollup
|
|
83
|
+
# digest in both windows (18,000 rows, 355ms on the rebulk environment's
|
|
84
|
+
# queries page) to produce four numbers that only move when RollupJob
|
|
85
|
+
# writes, which is at most once a minute per bucket; without this the
|
|
86
|
+
# merge ran again on every page load.
|
|
87
|
+
def self.summary_with_delta(environment, record_type, from:, to:, previous_from:, previous_to:, group_hash: nil)
|
|
88
|
+
key = [ "aggregations", "summary_with_delta", environment.id, record_type, group_hash,
|
|
89
|
+
from.to_i / 60, to.to_i / 60, previous_from.to_i / 60, previous_to.to_i / 60 ]
|
|
90
|
+
cached(key) do
|
|
91
|
+
environment.with_telemetry do
|
|
92
|
+
current = Telemetry::Rollup.for_type(record_type).between(from, to)
|
|
93
|
+
previous = Telemetry::Rollup.for_type(record_type).between(previous_from, previous_to)
|
|
94
|
+
if group_hash
|
|
95
|
+
current = current.where(group_hash: group_hash)
|
|
96
|
+
previous = previous.where(group_hash: group_hash)
|
|
97
|
+
end
|
|
98
|
+
{ current: Telemetry::Rollup.summarize(current), previous: Telemetry::Rollup.summarize(previous) }
|
|
99
|
+
end
|
|
100
|
+
end
|
|
101
|
+
end
|
|
102
|
+
|
|
103
|
+
# Hourly counts for a set of same-window rollup rows, zero-filled, coarsened
|
|
104
|
+
# to at most 30 points for wide windows (7d/30d).
|
|
105
|
+
def self.sparkline(rows, from, to)
|
|
106
|
+
hours = [ ((to - from) / 1.hour).ceil, 1 ].max
|
|
107
|
+
coarsen = (hours / 30.0).ceil
|
|
108
|
+
base = from.beginning_of_hour
|
|
109
|
+
buckets = Hash.new(0)
|
|
110
|
+
rows.each { |r| buckets[((r.bucket - base) / 1.hour).to_i / coarsen] += r.count }
|
|
111
|
+
bucket_count = (hours.to_f / coarsen).ceil
|
|
112
|
+
(0...bucket_count).map { |i| buckets[i] }
|
|
113
|
+
end
|
|
114
|
+
end
|
|
115
|
+
end
|
|
116
|
+
end
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Railwatch
|
|
4
|
+
module Telemetry
|
|
5
|
+
# A file an app attached to an execution or exception with Railwatch.attach.
|
|
6
|
+
# `data` is gzip of the original bytes.
|
|
7
|
+
class Attachment < TelemetryRecord
|
|
8
|
+
include Child
|
|
9
|
+
|
|
10
|
+
# A record has to fit inside the API's decoded request ceiling. Keeping
|
|
11
|
+
# the per-attachment ceiling equal to it preserves every valid existing
|
|
12
|
+
# payload while bounding reads of legacy rows.
|
|
13
|
+
MAX_BODY_BYTES = IngestRequestBodyLimit::MAX_BYTES
|
|
14
|
+
|
|
15
|
+
def body
|
|
16
|
+
BoundedGzip.decompress(data, max_bytes: MAX_BODY_BYTES)
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
# Only text-ish payloads are safe to render in the browser; everything
|
|
20
|
+
# else downloads. Rendered as text/plain either way, so a stored
|
|
21
|
+
# text/html attachment can never execute against our origin.
|
|
22
|
+
def viewable?
|
|
23
|
+
content_type.to_s.start_with?("text/") || content_type.to_s.split(";").first == "application/json"
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
def timeline_label
|
|
27
|
+
name.to_s.first(120)
|
|
28
|
+
end
|
|
29
|
+
end
|
|
30
|
+
end
|
|
31
|
+
end
|