railwatch 0.1.4 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +212 -0
- data/README.md +10 -0
- data/app/channels/railwatch/environment_channel.rb +28 -0
- data/app/controllers/concerns/railwatch/telemetry_identity.rb +25 -0
- data/app/controllers/railwatch/alerts_controller.rb +48 -0
- data/app/controllers/railwatch/anomaly_rules_controller.rb +31 -0
- data/app/controllers/railwatch/attachments_controller.rb +22 -0
- data/app/controllers/railwatch/beacon_controller.rb +67 -7
- data/app/controllers/railwatch/broadcasts_controller.rb +15 -0
- data/app/controllers/railwatch/cache_events_controller.rb +30 -0
- data/app/controllers/railwatch/commands_controller.rb +16 -0
- data/app/controllers/railwatch/comments_controller.rb +11 -0
- data/app/controllers/railwatch/dashboard_controller.rb +69 -0
- data/app/controllers/railwatch/deploys_controller.rb +34 -0
- data/app/controllers/railwatch/deprecations_controller.rb +54 -0
- data/app/controllers/railwatch/environment_scoped.rb +99 -0
- data/app/controllers/railwatch/exceptions_controller.rb +53 -0
- data/app/controllers/railwatch/executions_controller.rb +13 -0
- data/app/controllers/railwatch/issues_controller.rb +258 -0
- data/app/controllers/railwatch/jobs_controller.rb +63 -0
- data/app/controllers/railwatch/llm_calls_controller.rb +124 -0
- data/app/controllers/railwatch/logs_controller.rb +40 -0
- data/app/controllers/railwatch/mails_controller.rb +19 -0
- data/app/controllers/railwatch/notifications_controller.rb +11 -0
- data/app/controllers/railwatch/outgoing_requests_controller.rb +19 -0
- data/app/controllers/railwatch/overview_controller.rb +38 -0
- data/app/controllers/railwatch/people_controller.rb +25 -0
- data/app/controllers/railwatch/processes_controller.rb +33 -0
- data/app/controllers/railwatch/profiles_controller.rb +50 -0
- data/app/controllers/railwatch/queries_controller.rb +59 -0
- data/app/controllers/railwatch/releases_controller.rb +75 -0
- data/app/controllers/railwatch/requests_controller.rb +55 -0
- data/app/controllers/railwatch/saved_views_controller.rb +53 -0
- data/app/controllers/railwatch/scheduled_tasks_controller.rb +45 -0
- data/app/controllers/railwatch/spans_controller.rb +47 -0
- data/app/controllers/railwatch/storage_ops_controller.rb +15 -0
- data/app/controllers/railwatch/tenants_controller.rb +44 -0
- data/app/controllers/railwatch/thresholds_controller.rb +40 -0
- data/app/controllers/railwatch/traces_controller.rb +72 -0
- data/app/controllers/railwatch/transactions_controller.rb +21 -0
- data/app/controllers/railwatch/view_renders_controller.rb +28 -0
- data/app/controllers/railwatch/visits_controller.rb +53 -0
- data/app/helpers/railwatch/assets_helper.rb +52 -0
- data/app/jobs/railwatch/anomaly_scan_job.rb +11 -0
- data/app/jobs/railwatch/application_job.rb +7 -0
- data/app/jobs/railwatch/auto_resolve_issues_job.rb +19 -0
- data/app/jobs/railwatch/check_scheduled_tasks_job.rb +113 -0
- data/app/jobs/railwatch/detect_anomalies_job.rb +171 -0
- data/app/jobs/railwatch/detect_performance_issues_job.rb +85 -0
- data/app/jobs/railwatch/group_exceptions_job.rb +91 -0
- data/app/jobs/railwatch/optimize_telemetry_job.rb +23 -0
- data/app/jobs/railwatch/performance_scan_job.rb +11 -0
- data/app/jobs/railwatch/prune_telemetry_job.rb +110 -0
- data/app/jobs/railwatch/release_health_rollup_job.rb +64 -0
- data/app/jobs/railwatch/rollup_catchup_job.rb +21 -0
- data/app/jobs/railwatch/rollup_job.rb +130 -0
- data/app/jobs/railwatch/scheduled_task_scan_job.rb +11 -0
- data/app/models/concerns/railwatch/detection_snapshotting.rb +20 -0
- data/app/models/railwatch/alert.rb +229 -0
- data/app/models/railwatch/alert_rule.rb +121 -0
- data/app/models/railwatch/anomaly_rule.rb +33 -0
- data/app/models/railwatch/application.rb +44 -0
- data/app/models/railwatch/application_record.rb +33 -0
- data/app/models/railwatch/comment.rb +36 -0
- data/app/models/railwatch/deploy.rb +67 -0
- data/app/models/railwatch/environment.rb +54 -0
- data/app/models/railwatch/execution_presenter.rb +185 -0
- data/app/models/railwatch/filter_query.rb +143 -0
- data/app/models/railwatch/followup_receipt.rb +27 -0
- data/app/models/railwatch/ingest/batch.rb +305 -0
- data/app/models/railwatch/ingest/mapper.rb +575 -0
- data/app/models/railwatch/ingest/payload.rb +96 -0
- data/app/models/railwatch/ingest/rollup_absorber.rb +170 -0
- data/app/models/railwatch/ingest/writer.rb +137 -0
- data/app/models/railwatch/issue.rb +277 -0
- data/app/models/railwatch/issue_activity.rb +23 -0
- data/app/models/railwatch/issue_detection_presenter.rb +309 -0
- data/app/models/railwatch/issue_detection_snapshot.rb +77 -0
- data/app/models/railwatch/maintenance_task.rb +53 -0
- data/app/models/railwatch/saved_view.rb +63 -0
- data/app/models/railwatch/telemetry/aggregations.rb +116 -0
- data/app/models/railwatch/telemetry/attachment.rb +31 -0
- data/app/models/railwatch/telemetry/bounded_gzip.rb +101 -0
- data/app/models/railwatch/telemetry/broadcast.rb +13 -0
- data/app/models/railwatch/telemetry/cache_event.rb +16 -0
- data/app/models/railwatch/telemetry/child.rb +53 -0
- data/app/models/railwatch/telemetry/cursor_page.rb +99 -0
- data/app/models/railwatch/telemetry/deprecation.rb +13 -0
- data/app/models/railwatch/telemetry/enqueued_job.rb +13 -0
- data/app/models/railwatch/telemetry/exception.rb +21 -0
- data/app/models/railwatch/telemetry/execution.rb +71 -0
- data/app/models/railwatch/telemetry/health_sample.rb +72 -0
- data/app/models/railwatch/telemetry/ingest_batch.rb +31 -0
- data/app/models/railwatch/telemetry/llm_call.rb +37 -0
- data/app/models/railwatch/telemetry/log.rb +85 -0
- data/app/models/railwatch/telemetry/mail.rb +13 -0
- data/app/models/railwatch/telemetry/n_plus_one.rb +92 -0
- data/app/models/railwatch/telemetry/notification.rb +13 -0
- data/app/models/railwatch/telemetry/outgoing_request.rb +13 -0
- data/app/models/railwatch/telemetry/person.rb +50 -0
- data/app/models/railwatch/telemetry/process.rb +10 -0
- data/app/models/railwatch/telemetry/profile.rb +37 -0
- data/app/models/railwatch/telemetry/query.rb +50 -0
- data/app/models/railwatch/telemetry/query_shape.rb +45 -0
- data/app/models/railwatch/telemetry/release_health.rb +63 -0
- data/app/models/railwatch/telemetry/rollup.rb +78 -0
- data/app/models/railwatch/telemetry/session.rb +24 -0
- data/app/models/railwatch/telemetry/span.rb +19 -0
- data/app/models/railwatch/telemetry/storage_op.rb +13 -0
- data/app/models/railwatch/telemetry/tenant.rb +221 -0
- data/app/models/railwatch/telemetry/transaction.rb +13 -0
- data/app/models/railwatch/telemetry/view_render.rb +13 -0
- data/app/models/railwatch/telemetry/visit.rb +24 -0
- data/app/models/railwatch/telemetry_record.rb +57 -0
- data/app/models/railwatch/threshold.rb +29 -0
- data/app/models/railwatch/user.rb +47 -0
- data/app/models/railwatch/viewer.rb +13 -0
- data/app/views/layouts/railwatch/dashboard.html.erb +25 -0
- data/config/routes.rb +59 -0
- data/db/railwatch_migrate/20260916000000_create_railwatch_tables.rb +151 -0
- data/db/railwatch_migrate/20260917000000_create_railwatch_maintenance_tasks.rb +18 -0
- data/db/railwatch_migrate/20260917120000_create_railwatch_followup_receipts.rb +20 -0
- data/db/railwatch_migrate/20260918120000_widen_host_user_ids.rb +71 -0
- data/db/railwatch_telemetry_migrate/20260903000001_create_telemetry.rb +481 -0
- data/db/railwatch_telemetry_migrate/20260903000002_rename_tenant_to_app_tenant.rb +14 -0
- data/db/railwatch_telemetry_migrate/20260903000003_add_statement_count_to_transactions.rb +9 -0
- data/db/railwatch_telemetry_migrate/20260903000004_add_role_and_channel.rb +10 -0
- data/db/railwatch_telemetry_migrate/20260903000005_add_locals_to_exceptions.rb +9 -0
- data/db/railwatch_telemetry_migrate/20260903000006_add_spans_health_vitals_and_fts.rb +82 -0
- data/db/railwatch_telemetry_migrate/20260903000007_rename_span_attributes_to_payload.rb +10 -0
- data/db/railwatch_telemetry_migrate/20260903000008_add_profiles_and_attachments.rb +60 -0
- data/db/railwatch_telemetry_migrate/20260903000009_add_truncated_to_attachments.rb +9 -0
- data/db/railwatch_telemetry_migrate/20260903000010_create_sessions_and_release_health.rb +51 -0
- data/db/railwatch_telemetry_migrate/20260903000011_add_fingerprint_to_exceptions.rb +11 -0
- data/db/railwatch_telemetry_migrate/20260904000012_add_failed_to_broadcasts.rb +10 -0
- data/db/railwatch_telemetry_migrate/20260904010000_add_filter_cursor_indexes.rb +37 -0
- data/db/railwatch_telemetry_migrate/20260904020000_add_n_plus_ones_execution_id_index.rb +11 -0
- data/db/railwatch_telemetry_migrate/20260904120000_create_query_shapes.rb +15 -0
- data/db/railwatch_telemetry_migrate/20260906120000_add_backpressure_factor_to_ingest_batches.rb +9 -0
- data/db/railwatch_telemetry_migrate/20260907000000_rename_lantern_version_on_processes.rb +10 -0
- data/db/railwatch_telemetry_migrate/20260913000000_rename_nightrail_version_on_processes.rb +9 -0
- data/db/railwatch_telemetry_migrate/20260914000000_drop_orphan_durable_ingest_tables.rb +18 -0
- data/db/railwatch_telemetry_migrate/20260914010000_drop_orphan_durable_ingest_columns.rb +25 -0
- data/db/railwatch_telemetry_migrate/20260915000000_create_llm_calls.rb +55 -0
- data/db/railwatch_telemetry_migrate/20260915120000_add_detail_to_llm_calls.rb +21 -0
- data/db/railwatch_telemetry_migrate/20260915200000_add_explained_index_to_queries.rb +16 -0
- data/db/railwatch_telemetry_migrate/20260915210000_add_slowest_index_to_queries.rb +18 -0
- data/db/railwatch_telemetry_migrate/20260917010000_add_batch_ledger_to_ingest_batches.rb +16 -0
- data/docs/configuration.md +26 -0
- data/docs/embedded.md +352 -0
- data/docs/getting-started.md +5 -0
- data/lib/generators/railwatch/install/install_generator.rb +222 -1
- data/lib/generators/railwatch/install/templates/{initializer.rb → initializer.rb.tt} +39 -0
- data/lib/generators/railwatch/install/templates/post-deploy +8 -0
- data/lib/puma/plugin/railwatch.rb +170 -0
- data/lib/railwatch/authentication.rb +83 -0
- data/lib/railwatch/configuration.rb +134 -3
- data/lib/railwatch/dashboard_assets.rb +45 -0
- data/lib/railwatch/embedded.rb +54 -0
- data/lib/railwatch/engine.rb +89 -1
- data/lib/railwatch/ingest_request_body_limit.rb +10 -0
- data/lib/railwatch/json_compat.rb +60 -0
- data/lib/railwatch/maintenance.rb +183 -0
- data/lib/railwatch/patches/runner_command.rb +21 -1
- data/lib/railwatch/record.rb +33 -8
- data/lib/railwatch/reporter.rb +16 -0
- data/lib/railwatch/subscribers/process_info.rb +2 -1
- data/lib/railwatch/transport/local.rb +78 -0
- data/lib/railwatch/transport/socket.rb +183 -0
- data/lib/railwatch/version.rb +1 -1
- data/lib/railwatch/writer.rb +370 -0
- data/lib/railwatch.rb +38 -1
- data/lib/tasks/railwatch_tasks.rake +128 -0
- data/public/railwatch/assets/CommitMono-Bold-D6h61ieg.woff2 +0 -0
- data/public/railwatch/assets/CommitMono-Regular-zr8w7Obm.woff2 +0 -0
- data/public/railwatch/assets/Roboto-Black-auA4GeOK.woff2 +0 -0
- data/public/railwatch/assets/Roboto-Bold-CJLnO8j1.woff2 +0 -0
- data/public/railwatch/assets/Roboto-Medium-Cm2bwKpj.woff2 +0 -0
- data/public/railwatch/assets/Roboto-Regular-Chaq1-PV.woff2 +0 -0
- data/public/railwatch/assets/app-layout-yh-sPWgK.js +1 -0
- data/public/railwatch/assets/app-wordmark-nbjkzxwQ.js +1 -0
- data/public/railwatch/assets/appearance-CDuRvQTB.js +1 -0
- data/public/railwatch/assets/application-C_kpBdnf.css +1 -0
- data/public/railwatch/assets/arrow-up-DVtOGdVA.js +1 -0
- data/public/railwatch/assets/auth-layout-TCwPpS1K.js +1 -0
- data/public/railwatch/assets/badge-Daw4Hvr8.js +1 -0
- data/public/railwatch/assets/braces-DhFbHPsz.js +1 -0
- data/public/railwatch/assets/card-BX_3HXcJ.js +1 -0
- data/public/railwatch/assets/chart-B14-N9g7.js +39 -0
- data/public/railwatch/assets/chart-hover-Vy52H4uD.js +1 -0
- data/public/railwatch/assets/chart-panel-Cb4S_ej_.js +1 -0
- data/public/railwatch/assets/checkbox-D3aRSsBj.js +1 -0
- data/public/railwatch/assets/code-KvW8k7Jr.js +1 -0
- data/public/railwatch/assets/copy-DaQMJWoT.js +1 -0
- data/public/railwatch/assets/copy-block-BKFGd11J.js +1 -0
- data/public/railwatch/assets/copy-id-Djks1fXB.js +1 -0
- data/public/railwatch/assets/cursor-load-more-BXV0f1D_.js +1 -0
- data/public/railwatch/assets/data-table-iv1bdF6u.js +1 -0
- data/public/railwatch/assets/edit-BZc_Iawe.js +1 -0
- data/public/railwatch/assets/edit-DMKUF8Zi.js +8 -0
- data/public/railwatch/assets/edit-__9yJlO3.js +1 -0
- data/public/railwatch/assets/empty-state-6j_0AaQQ.js +1 -0
- data/public/railwatch/assets/env-layout-DrT8rO6P.js +1 -0
- data/public/railwatch/assets/execution-path-CzgBUi5e.js +1 -0
- data/public/railwatch/assets/filter-bar-9SU5NrzX.js +1 -0
- data/public/railwatch/assets/flamegraph-qcekju8V.js +2 -0
- data/public/railwatch/assets/format-B9SDkrWj.js +1 -0
- data/public/railwatch/assets/frames-Cyu7KMxZ.js +1 -0
- data/public/railwatch/assets/google-sign-in-button-DsTSfmzY.js +1 -0
- data/public/railwatch/assets/index-1ol1-QWI.js +1 -0
- data/public/railwatch/assets/index-5jI4aFzC.js +1 -0
- data/public/railwatch/assets/index-9KTrVnrc.js +1 -0
- data/public/railwatch/assets/index-B0-8lcTp.js +1 -0
- data/public/railwatch/assets/index-B7jjfNfO.js +1 -0
- data/public/railwatch/assets/index-BBchRy0M.js +1 -0
- data/public/railwatch/assets/index-BRiq3SNR.js +1 -0
- data/public/railwatch/assets/index-BgKj9xhr.js +1 -0
- data/public/railwatch/assets/index-BkTZqqOu.js +1 -0
- data/public/railwatch/assets/index-BprKx8QO.js +1 -0
- data/public/railwatch/assets/index-C3jzvPs3.js +1 -0
- data/public/railwatch/assets/index-Cbs6gGyQ.js +1 -0
- data/public/railwatch/assets/index-CdRZ6AWF.js +1 -0
- data/public/railwatch/assets/index-CeYKnapu.js +1 -0
- data/public/railwatch/assets/index-Cmlwy1-V.js +1 -0
- data/public/railwatch/assets/index-Cwx6058d.js +1 -0
- data/public/railwatch/assets/index-D4CSdbHv.js +1 -0
- data/public/railwatch/assets/index-DEFMSkdG.js +1 -0
- data/public/railwatch/assets/index-DFiHEBSh.js +1 -0
- data/public/railwatch/assets/index-DL4vWdWJ.js +1 -0
- data/public/railwatch/assets/index-DU9F5b5d.js +1 -0
- data/public/railwatch/assets/index-DaXgPcGL.js +1 -0
- data/public/railwatch/assets/index-DbtaU-EE.js +1 -0
- data/public/railwatch/assets/index-DeOe83F4.js +1 -0
- data/public/railwatch/assets/index-DiucHN4B.js +1 -0
- data/public/railwatch/assets/index-DlnR_l9o.js +1 -0
- data/public/railwatch/assets/index-DlumCsWY.js +2 -0
- data/public/railwatch/assets/index-DmRd7aIG.js +1 -0
- data/public/railwatch/assets/index-DxSh2UpM.js +1 -0
- data/public/railwatch/assets/index-MIMGuFNt.js +1 -0
- data/public/railwatch/assets/index-OqI59zPb.js +1 -0
- data/public/railwatch/assets/index-P4rC7IlX.js +1 -0
- data/public/railwatch/assets/index-gpPOcFWq.js +1 -0
- data/public/railwatch/assets/index-oVkururr.js +1 -0
- data/public/railwatch/assets/index-p9puqVge.js +1 -0
- data/public/railwatch/assets/inertia-TViv6kNv.js +97 -0
- data/public/railwatch/assets/input-error-LxImUkxv.js +1 -0
- data/public/railwatch/assets/json-viewer-Ar4cjPDW.js +1 -0
- data/public/railwatch/assets/klass-CJ-J4INB.js +1 -0
- data/public/railwatch/assets/label-COUKWqE_.js +1 -0
- data/public/railwatch/assets/layout-DNSLAkw_.js +1 -0
- data/public/railwatch/assets/live-dot-ChfUtY3p.js +41 -0
- data/public/railwatch/assets/nav-CNnDqPlm.js +1 -0
- data/public/railwatch/assets/new-84S8ZJq9.js +1 -0
- data/public/railwatch/assets/new-Be55nmt9.js +1 -0
- data/public/railwatch/assets/new-Bi_xQiIb.js +1 -0
- data/public/railwatch/assets/new-BvCT8TMg.js +1 -0
- data/public/railwatch/assets/new-D05SajFR.js +1 -0
- data/public/railwatch/assets/new-D4uewYC8.js +1 -0
- data/public/railwatch/assets/onboarding-CYZi5Cqc.js +1 -0
- data/public/railwatch/assets/origin-identity-6q1-CBts.js +1 -0
- data/public/railwatch/assets/percentile-picker-gFZCXtdb.js +1 -0
- data/public/railwatch/assets/relative-time-IOOgl5n2.js +1 -0
- data/public/railwatch/assets/release-health-DC8oc7uw.js +1 -0
- data/public/railwatch/assets/route-Dv6LAWvT.js +1 -0
- data/public/railwatch/assets/segmented-h1VdDTqE.js +1 -0
- data/public/railwatch/assets/select-_AJsUa7X.js +1 -0
- data/public/railwatch/assets/separator-BwwTYtCF.js +1 -0
- data/public/railwatch/assets/series-chart-DaFPefku.js +1 -0
- data/public/railwatch/assets/show-B7NCgkEo.js +1 -0
- data/public/railwatch/assets/show-BKqyKjBK.js +1 -0
- data/public/railwatch/assets/show-BM6X2Mpo.js +1 -0
- data/public/railwatch/assets/show-BNw4tN5q.js +1 -0
- data/public/railwatch/assets/show-BO3bnG5h.js +1 -0
- data/public/railwatch/assets/show-BhrAVAEA.js +1 -0
- data/public/railwatch/assets/show-C4Ltf5i9.js +2 -0
- data/public/railwatch/assets/show-C8sHalnw.js +1 -0
- data/public/railwatch/assets/show-CeTL4B37.js +2 -0
- data/public/railwatch/assets/show-CpfgV1jP.js +1 -0
- data/public/railwatch/assets/show-DACku6AD.js +3 -0
- data/public/railwatch/assets/show-DIOSGcXV.js +6 -0
- data/public/railwatch/assets/show-DQp_1n-B.js +1 -0
- data/public/railwatch/assets/show-DVNz46RI.js +1 -0
- data/public/railwatch/assets/show-DYteoYWW.js +1 -0
- data/public/railwatch/assets/show-DgSIoRvA.js +1 -0
- data/public/railwatch/assets/show-JxFtB4eK.js +2 -0
- data/public/railwatch/assets/sort-header-DpFzXblu.js +1 -0
- data/public/railwatch/assets/source-link-B2183i2-.js +1 -0
- data/public/railwatch/assets/sparkline-cell-C3-5vFkP.js +1 -0
- data/public/railwatch/assets/stat-s4RpOS9w.js +1 -0
- data/public/railwatch/assets/status-badge-8jVV-LA4.js +1 -0
- data/public/railwatch/assets/tenant-path-G-6u9A-o.js +1 -0
- data/public/railwatch/assets/text-link-DfsiaCcP.js +1 -0
- data/public/railwatch/assets/textarea-Dye72uP7.js +1 -0
- data/public/railwatch/assets/timeline-CD7WHnbo.js +1 -0
- data/public/railwatch/assets/transition-B_AW8rMK.js +5 -0
- data/public/railwatch/assets/use-clipboard-ColgLyQ2.js +1 -0
- data/public/railwatch/icon.png +0 -0
- data/public/railwatch/icon.svg +5 -0
- data/public/railwatch/manifest.json +2171 -0
- data/public/railwatch/rails-vite.json +1 -0
- metadata +314 -4
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "active_support/json"
|
|
4
|
+
|
|
5
|
+
module Railwatch
|
|
6
|
+
# Rails 8.1, up to and including 8.1.3.1, calls `JSON.parse` with a
|
|
7
|
+
# positional options hash. json 3 made those options keyword arguments
|
|
8
|
+
# (rails/rails#58784), so on that pair every signed cookie, every JSON
|
|
9
|
+
# column, and this gem's own SQLite migrations raise ArgumentError. Ruby
|
|
10
|
+
# 3.4.10 ships json 3.0.2 as a default gem, so a fresh machine meets it
|
|
11
|
+
# without choosing to. The fix is merged on Rails' 8-1-stable branch and
|
|
12
|
+
# unreleased as of 8.1.3.1.
|
|
13
|
+
#
|
|
14
|
+
# Detected by behaviour, not by version numbers: the pair is asked to
|
|
15
|
+
# decode a document once. That way the check is right about combinations
|
|
16
|
+
# nobody has enumerated (a patched Rails, a backport, a future json that
|
|
17
|
+
# restores the old signature), and it goes quiet by itself the day the
|
|
18
|
+
# host upgrades, with no release of this gem required.
|
|
19
|
+
#
|
|
20
|
+
# Railwatch does not pin `json` for the host. The breakage is the host
|
|
21
|
+
# application's either way -- its sessions are already failing -- so the
|
|
22
|
+
# gem reports it and the installer offers the pin, rather than quietly
|
|
23
|
+
# constraining everybody's bundle for a bug that is not ours and is on its
|
|
24
|
+
# way out.
|
|
25
|
+
module JsonCompat
|
|
26
|
+
PIN = %(gem "json", "< 3")
|
|
27
|
+
ISSUE = "rails/rails#58784"
|
|
28
|
+
|
|
29
|
+
module_function
|
|
30
|
+
|
|
31
|
+
def broken?
|
|
32
|
+
return @broken unless @broken.nil?
|
|
33
|
+
|
|
34
|
+
@broken = begin
|
|
35
|
+
ActiveSupport::JSON.decode("{}")
|
|
36
|
+
false
|
|
37
|
+
rescue ArgumentError
|
|
38
|
+
true
|
|
39
|
+
rescue StandardError
|
|
40
|
+
# Anything else is not this bug.
|
|
41
|
+
false
|
|
42
|
+
end
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
# One line, usable from the installer, the doctor and a boot log.
|
|
46
|
+
def advice
|
|
47
|
+
"json #{json_version} cannot be decoded by Rails #{rails_version} (#{ISSUE}): signed cookies, JSON " \
|
|
48
|
+
"columns and Railwatch's own migrations all raise ArgumentError on this pair. Add #{PIN} to your " \
|
|
49
|
+
"Gemfile until Rails ships the fix, then remove it."
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
def json_version = defined?(::JSON::VERSION) ? ::JSON::VERSION : "unknown"
|
|
53
|
+
def rails_version = defined?(::Rails) && ::Rails.respond_to?(:version) ? ::Rails.version : "unknown"
|
|
54
|
+
|
|
55
|
+
# Test hook.
|
|
56
|
+
def reset!
|
|
57
|
+
@broken = nil
|
|
58
|
+
end
|
|
59
|
+
end
|
|
60
|
+
end
|
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
# TASKS is built from duration literals at load time; the engine has these
|
|
4
|
+
# loaded already, but a writer process required ahead of the app must not
|
|
5
|
+
# depend on that.
|
|
6
|
+
require "active_support/core_ext/numeric/time"
|
|
7
|
+
|
|
8
|
+
module Railwatch
|
|
9
|
+
# The embedded install's own clock. One background thread per web/worker
|
|
10
|
+
# process wakes every TICK seconds and runs whichever maintenance task is
|
|
11
|
+
# due: release-health rollups, threshold and anomaly scans, missed
|
|
12
|
+
# scheduled tasks, auto-resolve, pruning. Every process runs the clock;
|
|
13
|
+
# one process at a time runs a given task, claimed through a lease row in
|
|
14
|
+
# the railwatch database (MaintenanceTask), so a Puma cluster and a Solid
|
|
15
|
+
# Queue worker on the same host do not all prune at once.
|
|
16
|
+
#
|
|
17
|
+
# Same thread shape as Railwatch::Health: parked on a ConditionVariable,
|
|
18
|
+
# started from the engine, re-armed in every forked child. Nothing here
|
|
19
|
+
# goes through Active Job -- the bodies call the job classes' perform
|
|
20
|
+
# directly -- so an embedded install needs no worker, no recurring.yml,
|
|
21
|
+
# and never writes the host's queue adapter.
|
|
22
|
+
#
|
|
23
|
+
# A task must never be visible to the app it is maintaining: each body runs
|
|
24
|
+
# inside Railwatch.ignore and rescues everything.
|
|
25
|
+
module Maintenance
|
|
26
|
+
ROLES = %w[web worker writer].freeze
|
|
27
|
+
TICK = 30
|
|
28
|
+
FOLLOWUP_BATCHES_PER_TICK = 200
|
|
29
|
+
|
|
30
|
+
# name => [interval, lease, body]. The lease is how long a claim is held
|
|
31
|
+
# by a process that never releases it (crashed mid-task); it is a ceiling
|
|
32
|
+
# on the work, not an estimate of it.
|
|
33
|
+
TASKS = {
|
|
34
|
+
# A batch's exceptions are grouped into issues right after it commits;
|
|
35
|
+
# a process that dies in between leaves the work recorded on the
|
|
36
|
+
# batch's ledger row. Bounded per tick so a long outage drains over a
|
|
37
|
+
# few ticks rather than one long one.
|
|
38
|
+
"drain_followups" => [ 1.minute, 5.minutes, lambda { |env|
|
|
39
|
+
env.with_telemetry do
|
|
40
|
+
Telemetry::IngestBatch.with_pending_followups.limit(FOLLOWUP_BATCHES_PER_TICK).each do |batch|
|
|
41
|
+
batch.drain_followups!(env)
|
|
42
|
+
end
|
|
43
|
+
end
|
|
44
|
+
} ],
|
|
45
|
+
"release_health" => [ 1.minute, 5.minutes, lambda { |env|
|
|
46
|
+
now = Time.current
|
|
47
|
+
[ now.beginning_of_hour, (now - 1.hour).beginning_of_hour ].each do |bucket|
|
|
48
|
+
ReleaseHealthRollupJob.new.perform(env, bucket)
|
|
49
|
+
end
|
|
50
|
+
} ],
|
|
51
|
+
# Embedded batches fold themselves into the current hour as they land
|
|
52
|
+
# (Ingest::RollupAbsorber), so the reconciler only ever needs to catch
|
|
53
|
+
# rows that arrived after their hour closed.
|
|
54
|
+
"rollup_reconcile" => [ 1.hour, 10.minutes, lambda { |env|
|
|
55
|
+
RollupJob.new.perform(env, (Time.current - 1.hour).beginning_of_hour)
|
|
56
|
+
} ],
|
|
57
|
+
"performance_scan" => [ 5.minutes, 10.minutes, lambda { |env|
|
|
58
|
+
DetectPerformanceIssuesJob.new.perform(env)
|
|
59
|
+
} ],
|
|
60
|
+
"anomaly_scan" => [ 5.minutes, 10.minutes, lambda { |env|
|
|
61
|
+
DetectAnomaliesJob.new.perform(env) if AnomalyRule.where(enabled: true).exists?
|
|
62
|
+
} ],
|
|
63
|
+
"scheduled_tasks" => [ 10.minutes, 10.minutes, lambda { |env|
|
|
64
|
+
CheckScheduledTasksJob.new.perform(env)
|
|
65
|
+
} ],
|
|
66
|
+
"auto_resolve" => [ 24.hours, 10.minutes, lambda { |_env|
|
|
67
|
+
AutoResolveIssuesJob.new.perform
|
|
68
|
+
} ],
|
|
69
|
+
# PASSIVE, not TRUNCATE: this runs inside a Puma worker, and a
|
|
70
|
+
# truncating checkpoint blocks every reader and writer on the file.
|
|
71
|
+
"prune" => [ 24.hours, 60.minutes, lambda { |env|
|
|
72
|
+
PruneTelemetryJob.new.perform(env, checkpoint: "PASSIVE")
|
|
73
|
+
OptimizeTelemetryJob.new.perform(env)
|
|
74
|
+
FollowupReceipt.prune!
|
|
75
|
+
} ]
|
|
76
|
+
}.freeze
|
|
77
|
+
|
|
78
|
+
@mutex = Mutex.new
|
|
79
|
+
@wakeup = ConditionVariable.new
|
|
80
|
+
@thread = nil
|
|
81
|
+
@pid = nil
|
|
82
|
+
@stopping = false
|
|
83
|
+
|
|
84
|
+
module_function
|
|
85
|
+
|
|
86
|
+
# Idempotent; Railwatch.restart_after_fork! calls it again in every forked
|
|
87
|
+
# child.
|
|
88
|
+
def start!
|
|
89
|
+
return unless Railwatch.enabled? && Railwatch.config.local?
|
|
90
|
+
return if defined?(Rails) && Rails.env.test?
|
|
91
|
+
return unless ROLES.include?(Subscribers::ProcessInfo.role)
|
|
92
|
+
# With a writer process listening, it is the one clock. The lease table
|
|
93
|
+
# would keep two clocks honest, but there is no reason to run a second.
|
|
94
|
+
return if !Writer.running? && Writer.listening?
|
|
95
|
+
return if @thread&.alive? && @pid == Process.pid
|
|
96
|
+
|
|
97
|
+
@mutex.synchronize do
|
|
98
|
+
return if @thread&.alive? && @pid == Process.pid
|
|
99
|
+
|
|
100
|
+
@pid = Process.pid
|
|
101
|
+
@stopping = false
|
|
102
|
+
@thread = Thread.new { run }
|
|
103
|
+
@thread.name = "railwatch-maintenance"
|
|
104
|
+
@thread.abort_on_exception = false
|
|
105
|
+
@thread.report_on_exception = false
|
|
106
|
+
end
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
# A forked child inherits a dead thread and may inherit a mutex held by a
|
|
110
|
+
# vanished parent thread; every primitive is replaced before start!.
|
|
111
|
+
def restart_after_fork!
|
|
112
|
+
@mutex = Mutex.new
|
|
113
|
+
@wakeup = ConditionVariable.new
|
|
114
|
+
@thread = nil
|
|
115
|
+
@pid = nil
|
|
116
|
+
@stopping = false
|
|
117
|
+
start!
|
|
118
|
+
end
|
|
119
|
+
|
|
120
|
+
def stop!
|
|
121
|
+
return unless @thread
|
|
122
|
+
|
|
123
|
+
@stopping = true
|
|
124
|
+
@mutex.synchronize { @wakeup.signal }
|
|
125
|
+
@thread.join(1)
|
|
126
|
+
@thread = nil
|
|
127
|
+
end
|
|
128
|
+
|
|
129
|
+
def run
|
|
130
|
+
until @stopping
|
|
131
|
+
@mutex.synchronize { @wakeup.wait(@mutex, TICK) unless @stopping }
|
|
132
|
+
tick unless @stopping
|
|
133
|
+
end
|
|
134
|
+
end
|
|
135
|
+
|
|
136
|
+
# Runs every task that is due and unclaimed. Returns the names it ran.
|
|
137
|
+
# Public so a spec, or an operator in a console, can drive the clock by
|
|
138
|
+
# hand. One failing task is reported and does not stop the others.
|
|
139
|
+
#
|
|
140
|
+
# A web worker's clock starts at boot, before the Puma plugin has forked
|
|
141
|
+
# the writer, so the start-time check in start! cannot see it. Checked
|
|
142
|
+
# again on every tick: once a writer is listening this process's clock
|
|
143
|
+
# stands down and stays down (the lease table would keep both honest,
|
|
144
|
+
# but there is no reason to run a second one).
|
|
145
|
+
def tick(now: Time.current)
|
|
146
|
+
return [] if !Writer.running? && Writer.listening?
|
|
147
|
+
|
|
148
|
+
ran = []
|
|
149
|
+
Rails.application.executor.wrap do
|
|
150
|
+
env = Environment.current
|
|
151
|
+
TASKS.each do |name, (every, lease, body)|
|
|
152
|
+
token = MaintenanceTask.claim(name, every: every, lease: lease, owner: owner, now: now) or next
|
|
153
|
+
|
|
154
|
+
succeeded = false
|
|
155
|
+
begin
|
|
156
|
+
Railwatch.ignore { body.call(env) }
|
|
157
|
+
succeeded = true
|
|
158
|
+
ran << name
|
|
159
|
+
rescue StandardError => e
|
|
160
|
+
# Not Rails.error: this thread has no execution for Railwatch.ignore
|
|
161
|
+
# to pause, so a report there would be captured by Railwatch's own
|
|
162
|
+
# subscriber and opened as an application issue about Railwatch.
|
|
163
|
+
Railwatch.debug { "maintenance #{name} failed: #{e.class}: #{e.message}" }
|
|
164
|
+
Railwatch.notify_unrecoverable(TaskError.new("maintenance task #{name} failed: #{e.class}: #{e.message}"))
|
|
165
|
+
ensure
|
|
166
|
+
MaintenanceTask.release(name, token: token, ran_at: now, succeeded: succeeded)
|
|
167
|
+
end
|
|
168
|
+
end
|
|
169
|
+
end
|
|
170
|
+
ran
|
|
171
|
+
rescue StandardError => e
|
|
172
|
+
# The lease table not being migrated yet is the usual way to get here.
|
|
173
|
+
Railwatch.debug { "maintenance tick failed: #{e.class}: #{e.message}" }
|
|
174
|
+
ran
|
|
175
|
+
end
|
|
176
|
+
|
|
177
|
+
def owner
|
|
178
|
+
"#{Railwatch.config.server}:#{Process.pid}"
|
|
179
|
+
end
|
|
180
|
+
|
|
181
|
+
class TaskError < StandardError; end
|
|
182
|
+
end
|
|
183
|
+
end
|
|
@@ -107,13 +107,33 @@ module Railwatch
|
|
|
107
107
|
# Expanded so a relative path is judged by where it actually resolves.
|
|
108
108
|
# An argument File.expand_path refuses (a "~nobody/x.rb") is matched
|
|
109
109
|
# as-is rather than assumed interactive: when in doubt, report.
|
|
110
|
+
#
|
|
111
|
+
# A scratch directory that contains the application itself is not a
|
|
112
|
+
# scratch directory for the application's own files: an app checked out
|
|
113
|
+
# under /tmp (a CI runner, a throwaway worktree) must not have its
|
|
114
|
+
# script/ classified as a shell session because of where the checkout
|
|
115
|
+
# lives. A scratch path inside the app (an operator's own
|
|
116
|
+
# `Rails.root/tmp/`) is left alone by this rule and still matches.
|
|
110
117
|
def self.scratch?(argument)
|
|
111
118
|
path = begin
|
|
112
119
|
File.expand_path(argument)
|
|
113
120
|
rescue StandardError
|
|
114
121
|
argument
|
|
115
122
|
end
|
|
116
|
-
|
|
123
|
+
root = application_root
|
|
124
|
+
Railwatch.config.interactive_runner_paths.any? do |directory|
|
|
125
|
+
next false if root && "#{root}/".start_with?(directory) && path.start_with?("#{root}/")
|
|
126
|
+
|
|
127
|
+
path.start_with?(directory)
|
|
128
|
+
end
|
|
129
|
+
end
|
|
130
|
+
|
|
131
|
+
def self.application_root
|
|
132
|
+
return nil unless defined?(Rails) && Rails.respond_to?(:root) && Rails.root
|
|
133
|
+
|
|
134
|
+
Rails.root.to_s
|
|
135
|
+
rescue StandardError
|
|
136
|
+
nil
|
|
117
137
|
end
|
|
118
138
|
end
|
|
119
139
|
end
|
data/lib/railwatch/record.rb
CHANGED
|
@@ -54,32 +54,57 @@ module Railwatch
|
|
|
54
54
|
# Counting stops as soon as `limit` is exceeded: past that the only fact
|
|
55
55
|
# the caller uses is "too big", so there is no reason to keep walking.
|
|
56
56
|
def buffered_bytes(value, limit:, depth: 0)
|
|
57
|
-
|
|
57
|
+
bytes = weigh(value, limit, depth)
|
|
58
|
+
bytes > limit ? limit + 1 : bytes
|
|
59
|
+
end
|
|
58
60
|
|
|
59
|
-
|
|
61
|
+
# The walk itself. Positional arguments, and scalars weighed inline in
|
|
62
|
+
# the container loops rather than through a call per value: a record is
|
|
63
|
+
# thirty-odd scalars under one Hash, so the common case is one call and
|
|
64
|
+
# one loop, and only a nested header or payload hash recurses. Halves the
|
|
65
|
+
# per-record cost against the one-method-per-value version (measured:
|
|
66
|
+
# 6.2 to 2.8 us for a query record).
|
|
67
|
+
def weigh(value, limit, depth)
|
|
68
|
+
case value
|
|
60
69
|
when String then 40 + value.bytesize
|
|
61
70
|
when Hash then hash_bytes(value, limit, depth)
|
|
62
71
|
when Array then array_bytes(value, limit, depth)
|
|
63
|
-
when Symbol then 16
|
|
64
72
|
else 16
|
|
65
73
|
end
|
|
66
|
-
bytes > limit ? limit + 1 : bytes
|
|
67
74
|
end
|
|
68
75
|
|
|
69
76
|
def hash_bytes(hash, limit, depth)
|
|
77
|
+
# Children are weighed inline below, so the bound is checked for them
|
|
78
|
+
# here: a container whose contents would sit past MAX_SIZING_DEPTH is
|
|
79
|
+
# over the limit by definition, which is what a per-value depth check
|
|
80
|
+
# produced before the contents were inlined.
|
|
81
|
+
return limit + 1 if depth > MAX_SIZING_DEPTH || (depth == MAX_SIZING_DEPTH && !hash.empty?)
|
|
82
|
+
|
|
70
83
|
bytes = 80 + (hash.size * 40)
|
|
71
|
-
hash.
|
|
72
|
-
bytes +=
|
|
73
|
-
bytes +=
|
|
84
|
+
hash.each_pair do |key, value|
|
|
85
|
+
bytes += key.is_a?(Symbol) ? 16 : weigh(key, limit, depth + 1)
|
|
86
|
+
bytes += case value
|
|
87
|
+
when String then 40 + value.bytesize
|
|
88
|
+
when Hash then hash_bytes(value, limit, depth + 1)
|
|
89
|
+
when Array then array_bytes(value, limit, depth + 1)
|
|
90
|
+
else 16
|
|
91
|
+
end
|
|
74
92
|
break if bytes > limit
|
|
75
93
|
end
|
|
76
94
|
bytes
|
|
77
95
|
end
|
|
78
96
|
|
|
79
97
|
def array_bytes(array, limit, depth)
|
|
98
|
+
return limit + 1 if depth > MAX_SIZING_DEPTH || (depth == MAX_SIZING_DEPTH && !array.empty?)
|
|
99
|
+
|
|
80
100
|
bytes = 40 + (array.size * 8)
|
|
81
101
|
array.each do |value|
|
|
82
|
-
bytes +=
|
|
102
|
+
bytes += case value
|
|
103
|
+
when String then 40 + value.bytesize
|
|
104
|
+
when Hash then hash_bytes(value, limit, depth + 1)
|
|
105
|
+
when Array then array_bytes(value, limit, depth + 1)
|
|
106
|
+
else 16
|
|
107
|
+
end
|
|
83
108
|
break if bytes > limit
|
|
84
109
|
end
|
|
85
110
|
bytes
|
data/lib/railwatch/reporter.rb
CHANGED
|
@@ -187,6 +187,12 @@ module Railwatch
|
|
|
187
187
|
end
|
|
188
188
|
|
|
189
189
|
def forked_transport
|
|
190
|
+
# A child may be a different kind of process from its parent: the
|
|
191
|
+
# writer forked from a Puma master must write SQLite itself, not hand
|
|
192
|
+
# batches back to the socket it is about to serve.
|
|
193
|
+
reselected = Railwatch.local_transport if @config.local?
|
|
194
|
+
return reselected if reselected && reselected.class != @transport.class
|
|
195
|
+
|
|
190
196
|
transport = @transport.dup
|
|
191
197
|
transport.reset_after_fork! if transport.respond_to?(:reset_after_fork!)
|
|
192
198
|
transport
|
|
@@ -327,6 +333,16 @@ module Railwatch
|
|
|
327
333
|
@retry_attempt = 0
|
|
328
334
|
@retry_at = nil
|
|
329
335
|
Railwatch.debug { "gave up on a batch of #{batch.records.size} records after #{MAX_RETRY_ATTEMPTS} retries (#{result.error || result.status}); dropped and counted" }
|
|
336
|
+
# Losing a batch is not a debug-level event: with an ingest (or an
|
|
337
|
+
# embedded writer) that never comes back this is the only place the
|
|
338
|
+
# loss is ever reported, and the dropped counter it leaves behind
|
|
339
|
+
# rides on the NEXT successful delivery, which may never happen.
|
|
340
|
+
Railwatch.notify_unrecoverable(
|
|
341
|
+
DeliveryError.new("Railwatch dropped #{batch.records.size} records after #{MAX_RETRY_ATTEMPTS} failed delivery attempts: " \
|
|
342
|
+
"#{result.error || result.status}",
|
|
343
|
+
status: result.status, records: batch.records.size, bytes: batch.bytes,
|
|
344
|
+
dropped: batch.dropped, dropped_bytes: batch.dropped_bytes)
|
|
345
|
+
)
|
|
330
346
|
next
|
|
331
347
|
end
|
|
332
348
|
@retry_batch = batch
|
|
@@ -89,7 +89,8 @@ module Railwatch
|
|
|
89
89
|
# $PROGRAM_NAME after boot and would otherwise turn the supervisor,
|
|
90
90
|
# dispatcher, and scheduler into "web" on every health sample.
|
|
91
91
|
def role
|
|
92
|
-
if defined?(::
|
|
92
|
+
if defined?(Railwatch::Writer) && Railwatch::Writer.running? then "writer"
|
|
93
|
+
elsif defined?(::SolidQueue) && ($PROGRAM_NAME.include?("jobs") || $PROGRAM_NAME.start_with?("solid-queue-") || ARGV.first.to_s.start_with?("solid_queue:")) then "worker"
|
|
93
94
|
elsif defined?(::Rails::Console) then "console"
|
|
94
95
|
elsif $PROGRAM_NAME.end_with?("rake") then "command"
|
|
95
96
|
elsif defined?(::Puma) then "web"
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Railwatch
|
|
4
|
+
module Transport
|
|
5
|
+
# Writes each batch straight into the engine's telemetry database instead
|
|
6
|
+
# of POSTing it. Same Reporter, same buffer, same backpressure; the only
|
|
7
|
+
# difference from Transport::Http is that deliver ends in Ingest::Batch
|
|
8
|
+
# rather than Net::HTTP. Runs on the reporter thread, never on a request.
|
|
9
|
+
class Local
|
|
10
|
+
Result = Struct.new(:ok, :status, :accepted, :rejected, :rejections, :error, :retryable_error, keyword_init: true) do
|
|
11
|
+
def retryable? = retryable_error ? true : false
|
|
12
|
+
end
|
|
13
|
+
|
|
14
|
+
def initialize(config)
|
|
15
|
+
@config = config
|
|
16
|
+
end
|
|
17
|
+
|
|
18
|
+
def deliver(records, dropped: 0, dropped_bytes: 0, backpressure_factor: 1.0, batch_id: nil)
|
|
19
|
+
wire = stringify(records)
|
|
20
|
+
# The reporter thread is not a request or a job: nothing has set up
|
|
21
|
+
# an ExecutionContext for it. Query log tags, Rails.error.report and
|
|
22
|
+
# connection checkin all assume one, so run the write inside the
|
|
23
|
+
# executor the same way a job would.
|
|
24
|
+
#
|
|
25
|
+
# The rescue is INSIDE the wrap. The executor reports any exception
|
|
26
|
+
# that escapes its block to Rails.error before re-raising it, and
|
|
27
|
+
# Railwatch subscribes to Rails.error: a write that failed (the
|
|
28
|
+
# telemetry database not migrated yet, a locked file) would otherwise
|
|
29
|
+
# be captured as one of the application's own exceptions and opened
|
|
30
|
+
# as an issue about Railwatch, by Railwatch.
|
|
31
|
+
Rails.application.executor.wrap do
|
|
32
|
+
write(wire, records.size, dropped: dropped, backpressure_factor: backpressure_factor, batch_id: batch_id)
|
|
33
|
+
rescue StandardError => e
|
|
34
|
+
Railwatch.debug { "local ingest failed: #{e.class}: #{e.message}" }
|
|
35
|
+
# With a batch id the reporter can safely retry: the ledger says
|
|
36
|
+
# whether the write committed, and the unique execution_id index
|
|
37
|
+
# makes a replay of a half-visible batch a no-op. Without one (an
|
|
38
|
+
# older caller) the batch is consumed, since a retry could
|
|
39
|
+
# double-insert. A malformed record never gets here: the mapper
|
|
40
|
+
# counts it as rejected and the rest of the batch is written.
|
|
41
|
+
Result.new(ok: false, error: "#{e.class}: #{e.message}", retryable_error: !batch_id.nil?)
|
|
42
|
+
end
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
def ping = true
|
|
46
|
+
def unauthorized? = false
|
|
47
|
+
def reset_after_fork! = self
|
|
48
|
+
|
|
49
|
+
private
|
|
50
|
+
|
|
51
|
+
def write(wire, count, dropped:, backpressure_factor:, batch_id:)
|
|
52
|
+
environment = Environment.current
|
|
53
|
+
# The same contract the HTTP path gets from X-Railwatch-Batch-Id: a
|
|
54
|
+
# batch the reporter retries after a failure is written once. The
|
|
55
|
+
# ledger row is created inside the batch transaction, so its presence
|
|
56
|
+
# is the commit.
|
|
57
|
+
if (ledger = environment.with_telemetry { Ingest::Batch.committed(batch_id) })
|
|
58
|
+
return Result.new(ok: true, status: 200, accepted: ledger.accepted, rejected: ledger.rejected, rejections: [])
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
result = Ingest::Batch.new(environment, wire, dropped_by_client: dropped, backpressure_factor: backpressure_factor,
|
|
62
|
+
gem_version: Railwatch::VERSION, embedded: true, batch_id: batch_id).write!
|
|
63
|
+
Result.new(ok: true, status: 200, accepted: result.accepted, rejected: result.rejected,
|
|
64
|
+
rejections: result.rejections.first(10))
|
|
65
|
+
end
|
|
66
|
+
|
|
67
|
+
# The gem builds symbol-keyed records, nested hashes included (stages,
|
|
68
|
+
# counters, headers); the mapper reads string keys throughout and
|
|
69
|
+
# rejects a symbol-keyed structure. A JSON round trip is the deep
|
|
70
|
+
# stringify. Measured against the alternatives on a 500-record batch:
|
|
71
|
+
# 1.9 ms and 11k allocations here versus 4.8 ms and 25k for
|
|
72
|
+
# deep_transform_keys, which also leaves Symbol values as Symbols.
|
|
73
|
+
def stringify(records)
|
|
74
|
+
JSON.parse(JSON.generate(records))
|
|
75
|
+
end
|
|
76
|
+
end
|
|
77
|
+
end
|
|
78
|
+
end
|
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "socket"
|
|
4
|
+
require "zlib"
|
|
5
|
+
require "json"
|
|
6
|
+
|
|
7
|
+
module Railwatch
|
|
8
|
+
module Transport
|
|
9
|
+
# Hands each batch to the embedded writer process (Railwatch::Writer) over
|
|
10
|
+
# a Unix socket instead of writing it into SQLite on this thread. The
|
|
11
|
+
# workers' Ruby threads then never map a record, hold SQLite's write lock,
|
|
12
|
+
# or run a t-digest; all of that happens in one process whose GVL no
|
|
13
|
+
# request shares. Same Reporter, same buffer, same batch id, same ledger:
|
|
14
|
+
# a batch the writer already committed replays as a no-op there.
|
|
15
|
+
#
|
|
16
|
+
# One request per connection, length-prefixed, gzip JSON both ways:
|
|
17
|
+
# > [u32 length][gzip(JSON {batch_id, records, dropped, dropped_bytes, backpressure_factor})]
|
|
18
|
+
# < [u32 length][JSON {ok, status, accepted, rejected, rejections, error, retryable_error}]
|
|
19
|
+
#
|
|
20
|
+
# Whether a writer is EXPECTED decides what a missing one means. Under
|
|
21
|
+
# the Puma plugin (which sets Writer.expected! in the master before it
|
|
22
|
+
# forks the workers) a socket that is absent or refusing is a writer
|
|
23
|
+
# that is starting or restarting: the batch is retained and retried on
|
|
24
|
+
# the reporter's backoff ladder, which is bounded -- after
|
|
25
|
+
# Reporter::MAX_RETRY_ATTEMPTS the batch is counted as dropped and
|
|
26
|
+
# reported through on_unrecoverable, rather than held for ever. Anywhere else
|
|
27
|
+
# (a plain `rails server`, a runner, a Solid Queue worker, the suite) no
|
|
28
|
+
# writer will ever appear, so the first miss switches this transport to
|
|
29
|
+
# an in-process Transport::Local for the rest of the process's life and
|
|
30
|
+
# embedded mode works exactly as it did before the writer existed. A
|
|
31
|
+
# refusing socket in that case is a stale inode from a dead writer and
|
|
32
|
+
# falls back the same way rather than dropping batches forever.
|
|
33
|
+
class Socket
|
|
34
|
+
Result = Local::Result
|
|
35
|
+
MAX_REPLY_BYTES = 1 << 20
|
|
36
|
+
# How long an in-process fallback lasts before the socket is tried
|
|
37
|
+
# again. The fallback is provisional on purpose: a process that missed
|
|
38
|
+
# the writer once (a bare `puma` whose plugin could not mark one
|
|
39
|
+
# expected, a worker forked before the writer bound) would otherwise
|
|
40
|
+
# write its own batches for the rest of its life.
|
|
41
|
+
FALLBACK_RECHECK = 30
|
|
42
|
+
# How long a process that EXPECTS a writer keeps retaining batches for
|
|
43
|
+
# one that never answers before it writes them itself. A writer that is
|
|
44
|
+
# restarting is back in seconds; one that cannot bind at all (an
|
|
45
|
+
# unwritable socket directory, a fork that keeps failing) would
|
|
46
|
+
# otherwise retain until the reporter's retry cap and then drop the
|
|
47
|
+
# batch. Writing it here instead keeps the records.
|
|
48
|
+
WRITER_GRACE = 60
|
|
49
|
+
|
|
50
|
+
def initialize(config, path: nil, expected: nil)
|
|
51
|
+
@config = config
|
|
52
|
+
@path = path || config.writer_socket_path
|
|
53
|
+
@expected = expected
|
|
54
|
+
# A path the kernel cannot bind (over 108 bytes on Linux) can never
|
|
55
|
+
# have a writer behind it; do not spend a batch finding out, and do
|
|
56
|
+
# not keep re-checking it either.
|
|
57
|
+
@unusable_path = !Writer.usable_path?(@path)
|
|
58
|
+
@fallback = @unusable_path ? Local.new(config) : nil
|
|
59
|
+
@fallback_at = Clock.monotonic if @fallback
|
|
60
|
+
Railwatch.debug { "writer socket path #{@path.inspect} is too long; writing batches in-process" } if @fallback
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
attr_reader :path
|
|
64
|
+
|
|
65
|
+
def fallback? = !@fallback.nil?
|
|
66
|
+
|
|
67
|
+
# Read when a batch misses, not when the transport is built: under
|
|
68
|
+
# `rails server` the app boots (and the reporter picks this transport)
|
|
69
|
+
# before Puma evaluates config/puma.rb, so the plugin's Writer.expected!
|
|
70
|
+
# lands after construction, and the workers inherit a copy of this
|
|
71
|
+
# object at fork. A flag captured here at construction was always false
|
|
72
|
+
# in every worker, and the first flush before the writer was listening
|
|
73
|
+
# fell back to in-process writes for the life of the worker.
|
|
74
|
+
def expected? = @expected.nil? ? Writer.expected? : @expected
|
|
75
|
+
|
|
76
|
+
def deliver(records, dropped: 0, dropped_bytes: 0, backpressure_factor: 1.0, batch_id: nil)
|
|
77
|
+
return @fallback.deliver(records, dropped: dropped, dropped_bytes: dropped_bytes,
|
|
78
|
+
backpressure_factor: backpressure_factor, batch_id: batch_id) if falling_back?
|
|
79
|
+
|
|
80
|
+
payload = Zlib.gzip(JSON.generate(batch_id: batch_id, records: records, dropped: dropped,
|
|
81
|
+
dropped_bytes: dropped_bytes, backpressure_factor: backpressure_factor))
|
|
82
|
+
reply = exchange(payload)
|
|
83
|
+
Result.new(**reply.slice("ok", "status", "accepted", "rejected", "rejections", "error", "retryable_error").transform_keys(&:to_sym))
|
|
84
|
+
rescue Errno::ENOENT, Errno::ECONNREFUSED, Errno::ENOTSOCK => e
|
|
85
|
+
@missing_since ||= Clock.monotonic
|
|
86
|
+
if expected? && Clock.monotonic - @missing_since < WRITER_GRACE
|
|
87
|
+
Railwatch.debug { "writer not answering at #{@path} (#{e.class}); retaining the batch" }
|
|
88
|
+
return Result.new(ok: false, error: "writer unavailable: #{e.message}", retryable_error: true)
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
Railwatch.debug do
|
|
92
|
+
if expected?
|
|
93
|
+
"no writer at #{@path} after #{WRITER_GRACE}s (#{e.class}); writing batches in-process, re-checking every #{FALLBACK_RECHECK}s"
|
|
94
|
+
else
|
|
95
|
+
"no writer at #{@path} (#{e.class}) and none expected; writing batches in-process, re-checking every #{FALLBACK_RECHECK}s"
|
|
96
|
+
end
|
|
97
|
+
end
|
|
98
|
+
@fallback = Local.new(@config)
|
|
99
|
+
@fallback_at = Clock.monotonic
|
|
100
|
+
deliver(records, dropped: dropped, dropped_bytes: dropped_bytes, backpressure_factor: backpressure_factor, batch_id: batch_id)
|
|
101
|
+
rescue SystemCallError, IOError, Zlib::Error, JSON::ParserError, Timeout::Error => e
|
|
102
|
+
Railwatch.debug { "writer delivery failed: #{e.class}: #{e.message}" }
|
|
103
|
+
Result.new(ok: false, error: "#{e.class}: #{e.message}", retryable_error: true)
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
# Whether this batch goes in-process. A fallback for an unbindable path
|
|
107
|
+
# is permanent; any other one is re-examined every FALLBACK_RECHECK
|
|
108
|
+
# seconds, so a writer that turns up later takes the work back.
|
|
109
|
+
def falling_back?
|
|
110
|
+
return false if @fallback.nil?
|
|
111
|
+
return true if @unusable_path || Clock.monotonic - @fallback_at < FALLBACK_RECHECK
|
|
112
|
+
|
|
113
|
+
if Writer.listening?(@path)
|
|
114
|
+
Railwatch.debug { "writer is answering at #{@path} again; handing batches back to it" }
|
|
115
|
+
@fallback = nil
|
|
116
|
+
@missing_since = nil
|
|
117
|
+
false
|
|
118
|
+
else
|
|
119
|
+
@fallback_at = Clock.monotonic
|
|
120
|
+
true
|
|
121
|
+
end
|
|
122
|
+
end
|
|
123
|
+
|
|
124
|
+
def ping
|
|
125
|
+
return true if @fallback
|
|
126
|
+
|
|
127
|
+
UNIXSocket.new(@path).close
|
|
128
|
+
true
|
|
129
|
+
rescue SystemCallError, ArgumentError
|
|
130
|
+
false
|
|
131
|
+
end
|
|
132
|
+
|
|
133
|
+
def unauthorized? = false
|
|
134
|
+
def reset_after_fork! = self
|
|
135
|
+
|
|
136
|
+
private
|
|
137
|
+
|
|
138
|
+
# Every read and write is under one deadline of config.timeout, the
|
|
139
|
+
# same budget the HTTP transport gives a POST; a writer that accepts but
|
|
140
|
+
# stops reading cannot hold the reporter thread past it.
|
|
141
|
+
def exchange(payload)
|
|
142
|
+
UNIXSocket.open(@path) do |sock|
|
|
143
|
+
deadline = Clock.monotonic + @config.timeout
|
|
144
|
+
write_exactly(sock, [ payload.bytesize ].pack("N") + payload, deadline)
|
|
145
|
+
header = read_exactly(sock, 4, deadline)
|
|
146
|
+
length = header.unpack1("N")
|
|
147
|
+
raise IOError, "writer reply too large (#{length} bytes)" if length > MAX_REPLY_BYTES
|
|
148
|
+
|
|
149
|
+
JSON.parse(read_exactly(sock, length, deadline))
|
|
150
|
+
end
|
|
151
|
+
end
|
|
152
|
+
|
|
153
|
+
def write_exactly(sock, data, deadline)
|
|
154
|
+
offset = 0
|
|
155
|
+
while offset < data.bytesize
|
|
156
|
+
remaining = deadline - Clock.monotonic
|
|
157
|
+
raise Timeout::Error, "writer did not accept the batch within #{@config.timeout}s" if remaining <= 0
|
|
158
|
+
raise IOError, "writer closed the connection" unless sock.wait_writable(remaining)
|
|
159
|
+
|
|
160
|
+
written = sock.write_nonblock(data.byteslice(offset..), exception: false)
|
|
161
|
+
offset += written if written.is_a?(Integer)
|
|
162
|
+
end
|
|
163
|
+
end
|
|
164
|
+
|
|
165
|
+
def read_exactly(sock, count, deadline)
|
|
166
|
+
buffer = +""
|
|
167
|
+
while buffer.bytesize < count
|
|
168
|
+
remaining = deadline - Clock.monotonic
|
|
169
|
+
raise Timeout::Error, "writer did not answer within #{@config.timeout}s" if remaining <= 0
|
|
170
|
+
raise IOError, "writer closed the connection" unless sock.wait_readable(remaining)
|
|
171
|
+
|
|
172
|
+
chunk = sock.read_nonblock(count - buffer.bytesize, exception: false)
|
|
173
|
+
case chunk
|
|
174
|
+
when :wait_readable then next
|
|
175
|
+
when nil then raise IOError, "writer closed the connection"
|
|
176
|
+
else buffer << chunk
|
|
177
|
+
end
|
|
178
|
+
end
|
|
179
|
+
buffer
|
|
180
|
+
end
|
|
181
|
+
end
|
|
182
|
+
end
|
|
183
|
+
end
|
data/lib/railwatch/version.rb
CHANGED