railwatch 0.1.4 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (302) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +212 -0
  3. data/README.md +10 -0
  4. data/app/channels/railwatch/environment_channel.rb +28 -0
  5. data/app/controllers/concerns/railwatch/telemetry_identity.rb +25 -0
  6. data/app/controllers/railwatch/alerts_controller.rb +48 -0
  7. data/app/controllers/railwatch/anomaly_rules_controller.rb +31 -0
  8. data/app/controllers/railwatch/attachments_controller.rb +22 -0
  9. data/app/controllers/railwatch/beacon_controller.rb +67 -7
  10. data/app/controllers/railwatch/broadcasts_controller.rb +15 -0
  11. data/app/controllers/railwatch/cache_events_controller.rb +30 -0
  12. data/app/controllers/railwatch/commands_controller.rb +16 -0
  13. data/app/controllers/railwatch/comments_controller.rb +11 -0
  14. data/app/controllers/railwatch/dashboard_controller.rb +69 -0
  15. data/app/controllers/railwatch/deploys_controller.rb +34 -0
  16. data/app/controllers/railwatch/deprecations_controller.rb +54 -0
  17. data/app/controllers/railwatch/environment_scoped.rb +99 -0
  18. data/app/controllers/railwatch/exceptions_controller.rb +53 -0
  19. data/app/controllers/railwatch/executions_controller.rb +13 -0
  20. data/app/controllers/railwatch/issues_controller.rb +258 -0
  21. data/app/controllers/railwatch/jobs_controller.rb +63 -0
  22. data/app/controllers/railwatch/llm_calls_controller.rb +124 -0
  23. data/app/controllers/railwatch/logs_controller.rb +40 -0
  24. data/app/controllers/railwatch/mails_controller.rb +19 -0
  25. data/app/controllers/railwatch/notifications_controller.rb +11 -0
  26. data/app/controllers/railwatch/outgoing_requests_controller.rb +19 -0
  27. data/app/controllers/railwatch/overview_controller.rb +38 -0
  28. data/app/controllers/railwatch/people_controller.rb +25 -0
  29. data/app/controllers/railwatch/processes_controller.rb +33 -0
  30. data/app/controllers/railwatch/profiles_controller.rb +50 -0
  31. data/app/controllers/railwatch/queries_controller.rb +59 -0
  32. data/app/controllers/railwatch/releases_controller.rb +75 -0
  33. data/app/controllers/railwatch/requests_controller.rb +55 -0
  34. data/app/controllers/railwatch/saved_views_controller.rb +53 -0
  35. data/app/controllers/railwatch/scheduled_tasks_controller.rb +45 -0
  36. data/app/controllers/railwatch/spans_controller.rb +47 -0
  37. data/app/controllers/railwatch/storage_ops_controller.rb +15 -0
  38. data/app/controllers/railwatch/tenants_controller.rb +44 -0
  39. data/app/controllers/railwatch/thresholds_controller.rb +40 -0
  40. data/app/controllers/railwatch/traces_controller.rb +72 -0
  41. data/app/controllers/railwatch/transactions_controller.rb +21 -0
  42. data/app/controllers/railwatch/view_renders_controller.rb +28 -0
  43. data/app/controllers/railwatch/visits_controller.rb +53 -0
  44. data/app/helpers/railwatch/assets_helper.rb +52 -0
  45. data/app/jobs/railwatch/anomaly_scan_job.rb +11 -0
  46. data/app/jobs/railwatch/application_job.rb +7 -0
  47. data/app/jobs/railwatch/auto_resolve_issues_job.rb +19 -0
  48. data/app/jobs/railwatch/check_scheduled_tasks_job.rb +113 -0
  49. data/app/jobs/railwatch/detect_anomalies_job.rb +171 -0
  50. data/app/jobs/railwatch/detect_performance_issues_job.rb +85 -0
  51. data/app/jobs/railwatch/group_exceptions_job.rb +91 -0
  52. data/app/jobs/railwatch/optimize_telemetry_job.rb +23 -0
  53. data/app/jobs/railwatch/performance_scan_job.rb +11 -0
  54. data/app/jobs/railwatch/prune_telemetry_job.rb +110 -0
  55. data/app/jobs/railwatch/release_health_rollup_job.rb +64 -0
  56. data/app/jobs/railwatch/rollup_catchup_job.rb +21 -0
  57. data/app/jobs/railwatch/rollup_job.rb +130 -0
  58. data/app/jobs/railwatch/scheduled_task_scan_job.rb +11 -0
  59. data/app/models/concerns/railwatch/detection_snapshotting.rb +20 -0
  60. data/app/models/railwatch/alert.rb +229 -0
  61. data/app/models/railwatch/alert_rule.rb +121 -0
  62. data/app/models/railwatch/anomaly_rule.rb +33 -0
  63. data/app/models/railwatch/application.rb +44 -0
  64. data/app/models/railwatch/application_record.rb +33 -0
  65. data/app/models/railwatch/comment.rb +36 -0
  66. data/app/models/railwatch/deploy.rb +67 -0
  67. data/app/models/railwatch/environment.rb +54 -0
  68. data/app/models/railwatch/execution_presenter.rb +185 -0
  69. data/app/models/railwatch/filter_query.rb +143 -0
  70. data/app/models/railwatch/followup_receipt.rb +27 -0
  71. data/app/models/railwatch/ingest/batch.rb +305 -0
  72. data/app/models/railwatch/ingest/mapper.rb +575 -0
  73. data/app/models/railwatch/ingest/payload.rb +96 -0
  74. data/app/models/railwatch/ingest/rollup_absorber.rb +170 -0
  75. data/app/models/railwatch/ingest/writer.rb +137 -0
  76. data/app/models/railwatch/issue.rb +277 -0
  77. data/app/models/railwatch/issue_activity.rb +23 -0
  78. data/app/models/railwatch/issue_detection_presenter.rb +309 -0
  79. data/app/models/railwatch/issue_detection_snapshot.rb +77 -0
  80. data/app/models/railwatch/maintenance_task.rb +53 -0
  81. data/app/models/railwatch/saved_view.rb +63 -0
  82. data/app/models/railwatch/telemetry/aggregations.rb +116 -0
  83. data/app/models/railwatch/telemetry/attachment.rb +31 -0
  84. data/app/models/railwatch/telemetry/bounded_gzip.rb +101 -0
  85. data/app/models/railwatch/telemetry/broadcast.rb +13 -0
  86. data/app/models/railwatch/telemetry/cache_event.rb +16 -0
  87. data/app/models/railwatch/telemetry/child.rb +53 -0
  88. data/app/models/railwatch/telemetry/cursor_page.rb +99 -0
  89. data/app/models/railwatch/telemetry/deprecation.rb +13 -0
  90. data/app/models/railwatch/telemetry/enqueued_job.rb +13 -0
  91. data/app/models/railwatch/telemetry/exception.rb +21 -0
  92. data/app/models/railwatch/telemetry/execution.rb +71 -0
  93. data/app/models/railwatch/telemetry/health_sample.rb +72 -0
  94. data/app/models/railwatch/telemetry/ingest_batch.rb +31 -0
  95. data/app/models/railwatch/telemetry/llm_call.rb +37 -0
  96. data/app/models/railwatch/telemetry/log.rb +85 -0
  97. data/app/models/railwatch/telemetry/mail.rb +13 -0
  98. data/app/models/railwatch/telemetry/n_plus_one.rb +92 -0
  99. data/app/models/railwatch/telemetry/notification.rb +13 -0
  100. data/app/models/railwatch/telemetry/outgoing_request.rb +13 -0
  101. data/app/models/railwatch/telemetry/person.rb +50 -0
  102. data/app/models/railwatch/telemetry/process.rb +10 -0
  103. data/app/models/railwatch/telemetry/profile.rb +37 -0
  104. data/app/models/railwatch/telemetry/query.rb +50 -0
  105. data/app/models/railwatch/telemetry/query_shape.rb +45 -0
  106. data/app/models/railwatch/telemetry/release_health.rb +63 -0
  107. data/app/models/railwatch/telemetry/rollup.rb +78 -0
  108. data/app/models/railwatch/telemetry/session.rb +24 -0
  109. data/app/models/railwatch/telemetry/span.rb +19 -0
  110. data/app/models/railwatch/telemetry/storage_op.rb +13 -0
  111. data/app/models/railwatch/telemetry/tenant.rb +221 -0
  112. data/app/models/railwatch/telemetry/transaction.rb +13 -0
  113. data/app/models/railwatch/telemetry/view_render.rb +13 -0
  114. data/app/models/railwatch/telemetry/visit.rb +24 -0
  115. data/app/models/railwatch/telemetry_record.rb +57 -0
  116. data/app/models/railwatch/threshold.rb +29 -0
  117. data/app/models/railwatch/user.rb +47 -0
  118. data/app/models/railwatch/viewer.rb +13 -0
  119. data/app/views/layouts/railwatch/dashboard.html.erb +25 -0
  120. data/config/routes.rb +59 -0
  121. data/db/railwatch_migrate/20260916000000_create_railwatch_tables.rb +151 -0
  122. data/db/railwatch_migrate/20260917000000_create_railwatch_maintenance_tasks.rb +18 -0
  123. data/db/railwatch_migrate/20260917120000_create_railwatch_followup_receipts.rb +20 -0
  124. data/db/railwatch_migrate/20260918120000_widen_host_user_ids.rb +71 -0
  125. data/db/railwatch_telemetry_migrate/20260903000001_create_telemetry.rb +481 -0
  126. data/db/railwatch_telemetry_migrate/20260903000002_rename_tenant_to_app_tenant.rb +14 -0
  127. data/db/railwatch_telemetry_migrate/20260903000003_add_statement_count_to_transactions.rb +9 -0
  128. data/db/railwatch_telemetry_migrate/20260903000004_add_role_and_channel.rb +10 -0
  129. data/db/railwatch_telemetry_migrate/20260903000005_add_locals_to_exceptions.rb +9 -0
  130. data/db/railwatch_telemetry_migrate/20260903000006_add_spans_health_vitals_and_fts.rb +82 -0
  131. data/db/railwatch_telemetry_migrate/20260903000007_rename_span_attributes_to_payload.rb +10 -0
  132. data/db/railwatch_telemetry_migrate/20260903000008_add_profiles_and_attachments.rb +60 -0
  133. data/db/railwatch_telemetry_migrate/20260903000009_add_truncated_to_attachments.rb +9 -0
  134. data/db/railwatch_telemetry_migrate/20260903000010_create_sessions_and_release_health.rb +51 -0
  135. data/db/railwatch_telemetry_migrate/20260903000011_add_fingerprint_to_exceptions.rb +11 -0
  136. data/db/railwatch_telemetry_migrate/20260904000012_add_failed_to_broadcasts.rb +10 -0
  137. data/db/railwatch_telemetry_migrate/20260904010000_add_filter_cursor_indexes.rb +37 -0
  138. data/db/railwatch_telemetry_migrate/20260904020000_add_n_plus_ones_execution_id_index.rb +11 -0
  139. data/db/railwatch_telemetry_migrate/20260904120000_create_query_shapes.rb +15 -0
  140. data/db/railwatch_telemetry_migrate/20260906120000_add_backpressure_factor_to_ingest_batches.rb +9 -0
  141. data/db/railwatch_telemetry_migrate/20260907000000_rename_lantern_version_on_processes.rb +10 -0
  142. data/db/railwatch_telemetry_migrate/20260913000000_rename_nightrail_version_on_processes.rb +9 -0
  143. data/db/railwatch_telemetry_migrate/20260914000000_drop_orphan_durable_ingest_tables.rb +18 -0
  144. data/db/railwatch_telemetry_migrate/20260914010000_drop_orphan_durable_ingest_columns.rb +25 -0
  145. data/db/railwatch_telemetry_migrate/20260915000000_create_llm_calls.rb +55 -0
  146. data/db/railwatch_telemetry_migrate/20260915120000_add_detail_to_llm_calls.rb +21 -0
  147. data/db/railwatch_telemetry_migrate/20260915200000_add_explained_index_to_queries.rb +16 -0
  148. data/db/railwatch_telemetry_migrate/20260915210000_add_slowest_index_to_queries.rb +18 -0
  149. data/db/railwatch_telemetry_migrate/20260917010000_add_batch_ledger_to_ingest_batches.rb +16 -0
  150. data/docs/configuration.md +26 -0
  151. data/docs/embedded.md +352 -0
  152. data/docs/getting-started.md +5 -0
  153. data/lib/generators/railwatch/install/install_generator.rb +222 -1
  154. data/lib/generators/railwatch/install/templates/{initializer.rb → initializer.rb.tt} +39 -0
  155. data/lib/generators/railwatch/install/templates/post-deploy +8 -0
  156. data/lib/puma/plugin/railwatch.rb +170 -0
  157. data/lib/railwatch/authentication.rb +83 -0
  158. data/lib/railwatch/configuration.rb +134 -3
  159. data/lib/railwatch/dashboard_assets.rb +45 -0
  160. data/lib/railwatch/embedded.rb +54 -0
  161. data/lib/railwatch/engine.rb +89 -1
  162. data/lib/railwatch/ingest_request_body_limit.rb +10 -0
  163. data/lib/railwatch/json_compat.rb +60 -0
  164. data/lib/railwatch/maintenance.rb +183 -0
  165. data/lib/railwatch/patches/runner_command.rb +21 -1
  166. data/lib/railwatch/record.rb +33 -8
  167. data/lib/railwatch/reporter.rb +16 -0
  168. data/lib/railwatch/subscribers/process_info.rb +2 -1
  169. data/lib/railwatch/transport/local.rb +78 -0
  170. data/lib/railwatch/transport/socket.rb +183 -0
  171. data/lib/railwatch/version.rb +1 -1
  172. data/lib/railwatch/writer.rb +370 -0
  173. data/lib/railwatch.rb +38 -1
  174. data/lib/tasks/railwatch_tasks.rake +128 -0
  175. data/public/railwatch/assets/CommitMono-Bold-D6h61ieg.woff2 +0 -0
  176. data/public/railwatch/assets/CommitMono-Regular-zr8w7Obm.woff2 +0 -0
  177. data/public/railwatch/assets/Roboto-Black-auA4GeOK.woff2 +0 -0
  178. data/public/railwatch/assets/Roboto-Bold-CJLnO8j1.woff2 +0 -0
  179. data/public/railwatch/assets/Roboto-Medium-Cm2bwKpj.woff2 +0 -0
  180. data/public/railwatch/assets/Roboto-Regular-Chaq1-PV.woff2 +0 -0
  181. data/public/railwatch/assets/app-layout-yh-sPWgK.js +1 -0
  182. data/public/railwatch/assets/app-wordmark-nbjkzxwQ.js +1 -0
  183. data/public/railwatch/assets/appearance-CDuRvQTB.js +1 -0
  184. data/public/railwatch/assets/application-C_kpBdnf.css +1 -0
  185. data/public/railwatch/assets/arrow-up-DVtOGdVA.js +1 -0
  186. data/public/railwatch/assets/auth-layout-TCwPpS1K.js +1 -0
  187. data/public/railwatch/assets/badge-Daw4Hvr8.js +1 -0
  188. data/public/railwatch/assets/braces-DhFbHPsz.js +1 -0
  189. data/public/railwatch/assets/card-BX_3HXcJ.js +1 -0
  190. data/public/railwatch/assets/chart-B14-N9g7.js +39 -0
  191. data/public/railwatch/assets/chart-hover-Vy52H4uD.js +1 -0
  192. data/public/railwatch/assets/chart-panel-Cb4S_ej_.js +1 -0
  193. data/public/railwatch/assets/checkbox-D3aRSsBj.js +1 -0
  194. data/public/railwatch/assets/code-KvW8k7Jr.js +1 -0
  195. data/public/railwatch/assets/copy-DaQMJWoT.js +1 -0
  196. data/public/railwatch/assets/copy-block-BKFGd11J.js +1 -0
  197. data/public/railwatch/assets/copy-id-Djks1fXB.js +1 -0
  198. data/public/railwatch/assets/cursor-load-more-BXV0f1D_.js +1 -0
  199. data/public/railwatch/assets/data-table-iv1bdF6u.js +1 -0
  200. data/public/railwatch/assets/edit-BZc_Iawe.js +1 -0
  201. data/public/railwatch/assets/edit-DMKUF8Zi.js +8 -0
  202. data/public/railwatch/assets/edit-__9yJlO3.js +1 -0
  203. data/public/railwatch/assets/empty-state-6j_0AaQQ.js +1 -0
  204. data/public/railwatch/assets/env-layout-DrT8rO6P.js +1 -0
  205. data/public/railwatch/assets/execution-path-CzgBUi5e.js +1 -0
  206. data/public/railwatch/assets/filter-bar-9SU5NrzX.js +1 -0
  207. data/public/railwatch/assets/flamegraph-qcekju8V.js +2 -0
  208. data/public/railwatch/assets/format-B9SDkrWj.js +1 -0
  209. data/public/railwatch/assets/frames-Cyu7KMxZ.js +1 -0
  210. data/public/railwatch/assets/google-sign-in-button-DsTSfmzY.js +1 -0
  211. data/public/railwatch/assets/index-1ol1-QWI.js +1 -0
  212. data/public/railwatch/assets/index-5jI4aFzC.js +1 -0
  213. data/public/railwatch/assets/index-9KTrVnrc.js +1 -0
  214. data/public/railwatch/assets/index-B0-8lcTp.js +1 -0
  215. data/public/railwatch/assets/index-B7jjfNfO.js +1 -0
  216. data/public/railwatch/assets/index-BBchRy0M.js +1 -0
  217. data/public/railwatch/assets/index-BRiq3SNR.js +1 -0
  218. data/public/railwatch/assets/index-BgKj9xhr.js +1 -0
  219. data/public/railwatch/assets/index-BkTZqqOu.js +1 -0
  220. data/public/railwatch/assets/index-BprKx8QO.js +1 -0
  221. data/public/railwatch/assets/index-C3jzvPs3.js +1 -0
  222. data/public/railwatch/assets/index-Cbs6gGyQ.js +1 -0
  223. data/public/railwatch/assets/index-CdRZ6AWF.js +1 -0
  224. data/public/railwatch/assets/index-CeYKnapu.js +1 -0
  225. data/public/railwatch/assets/index-Cmlwy1-V.js +1 -0
  226. data/public/railwatch/assets/index-Cwx6058d.js +1 -0
  227. data/public/railwatch/assets/index-D4CSdbHv.js +1 -0
  228. data/public/railwatch/assets/index-DEFMSkdG.js +1 -0
  229. data/public/railwatch/assets/index-DFiHEBSh.js +1 -0
  230. data/public/railwatch/assets/index-DL4vWdWJ.js +1 -0
  231. data/public/railwatch/assets/index-DU9F5b5d.js +1 -0
  232. data/public/railwatch/assets/index-DaXgPcGL.js +1 -0
  233. data/public/railwatch/assets/index-DbtaU-EE.js +1 -0
  234. data/public/railwatch/assets/index-DeOe83F4.js +1 -0
  235. data/public/railwatch/assets/index-DiucHN4B.js +1 -0
  236. data/public/railwatch/assets/index-DlnR_l9o.js +1 -0
  237. data/public/railwatch/assets/index-DlumCsWY.js +2 -0
  238. data/public/railwatch/assets/index-DmRd7aIG.js +1 -0
  239. data/public/railwatch/assets/index-DxSh2UpM.js +1 -0
  240. data/public/railwatch/assets/index-MIMGuFNt.js +1 -0
  241. data/public/railwatch/assets/index-OqI59zPb.js +1 -0
  242. data/public/railwatch/assets/index-P4rC7IlX.js +1 -0
  243. data/public/railwatch/assets/index-gpPOcFWq.js +1 -0
  244. data/public/railwatch/assets/index-oVkururr.js +1 -0
  245. data/public/railwatch/assets/index-p9puqVge.js +1 -0
  246. data/public/railwatch/assets/inertia-TViv6kNv.js +97 -0
  247. data/public/railwatch/assets/input-error-LxImUkxv.js +1 -0
  248. data/public/railwatch/assets/json-viewer-Ar4cjPDW.js +1 -0
  249. data/public/railwatch/assets/klass-CJ-J4INB.js +1 -0
  250. data/public/railwatch/assets/label-COUKWqE_.js +1 -0
  251. data/public/railwatch/assets/layout-DNSLAkw_.js +1 -0
  252. data/public/railwatch/assets/live-dot-ChfUtY3p.js +41 -0
  253. data/public/railwatch/assets/nav-CNnDqPlm.js +1 -0
  254. data/public/railwatch/assets/new-84S8ZJq9.js +1 -0
  255. data/public/railwatch/assets/new-Be55nmt9.js +1 -0
  256. data/public/railwatch/assets/new-Bi_xQiIb.js +1 -0
  257. data/public/railwatch/assets/new-BvCT8TMg.js +1 -0
  258. data/public/railwatch/assets/new-D05SajFR.js +1 -0
  259. data/public/railwatch/assets/new-D4uewYC8.js +1 -0
  260. data/public/railwatch/assets/onboarding-CYZi5Cqc.js +1 -0
  261. data/public/railwatch/assets/origin-identity-6q1-CBts.js +1 -0
  262. data/public/railwatch/assets/percentile-picker-gFZCXtdb.js +1 -0
  263. data/public/railwatch/assets/relative-time-IOOgl5n2.js +1 -0
  264. data/public/railwatch/assets/release-health-DC8oc7uw.js +1 -0
  265. data/public/railwatch/assets/route-Dv6LAWvT.js +1 -0
  266. data/public/railwatch/assets/segmented-h1VdDTqE.js +1 -0
  267. data/public/railwatch/assets/select-_AJsUa7X.js +1 -0
  268. data/public/railwatch/assets/separator-BwwTYtCF.js +1 -0
  269. data/public/railwatch/assets/series-chart-DaFPefku.js +1 -0
  270. data/public/railwatch/assets/show-B7NCgkEo.js +1 -0
  271. data/public/railwatch/assets/show-BKqyKjBK.js +1 -0
  272. data/public/railwatch/assets/show-BM6X2Mpo.js +1 -0
  273. data/public/railwatch/assets/show-BNw4tN5q.js +1 -0
  274. data/public/railwatch/assets/show-BO3bnG5h.js +1 -0
  275. data/public/railwatch/assets/show-BhrAVAEA.js +1 -0
  276. data/public/railwatch/assets/show-C4Ltf5i9.js +2 -0
  277. data/public/railwatch/assets/show-C8sHalnw.js +1 -0
  278. data/public/railwatch/assets/show-CeTL4B37.js +2 -0
  279. data/public/railwatch/assets/show-CpfgV1jP.js +1 -0
  280. data/public/railwatch/assets/show-DACku6AD.js +3 -0
  281. data/public/railwatch/assets/show-DIOSGcXV.js +6 -0
  282. data/public/railwatch/assets/show-DQp_1n-B.js +1 -0
  283. data/public/railwatch/assets/show-DVNz46RI.js +1 -0
  284. data/public/railwatch/assets/show-DYteoYWW.js +1 -0
  285. data/public/railwatch/assets/show-DgSIoRvA.js +1 -0
  286. data/public/railwatch/assets/show-JxFtB4eK.js +2 -0
  287. data/public/railwatch/assets/sort-header-DpFzXblu.js +1 -0
  288. data/public/railwatch/assets/source-link-B2183i2-.js +1 -0
  289. data/public/railwatch/assets/sparkline-cell-C3-5vFkP.js +1 -0
  290. data/public/railwatch/assets/stat-s4RpOS9w.js +1 -0
  291. data/public/railwatch/assets/status-badge-8jVV-LA4.js +1 -0
  292. data/public/railwatch/assets/tenant-path-G-6u9A-o.js +1 -0
  293. data/public/railwatch/assets/text-link-DfsiaCcP.js +1 -0
  294. data/public/railwatch/assets/textarea-Dye72uP7.js +1 -0
  295. data/public/railwatch/assets/timeline-CD7WHnbo.js +1 -0
  296. data/public/railwatch/assets/transition-B_AW8rMK.js +5 -0
  297. data/public/railwatch/assets/use-clipboard-ColgLyQ2.js +1 -0
  298. data/public/railwatch/icon.png +0 -0
  299. data/public/railwatch/icon.svg +5 -0
  300. data/public/railwatch/manifest.json +2171 -0
  301. data/public/railwatch/rails-vite.json +1 -0
  302. metadata +314 -4
@@ -0,0 +1,171 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Railwatch
4
+ # Compares each enabled anomaly rule's current window against the same clock
5
+ # window on each of the previous `baseline_days` days and opens an anomaly
6
+ # issue when the current value sits above the baseline's mean plus
7
+ # `deviation` standard deviations. Runs on a schedule via AnomalyScanJob.
8
+ class DetectAnomaliesJob < ApplicationJob
9
+ include DetectionSnapshotting
10
+
11
+ queue_as :default
12
+
13
+ TYPE_FOR = DetectPerformanceIssuesJob::TYPE_FOR
14
+ # Fewer baseline days than this and the standard deviation is noise.
15
+ MIN_BASELINE_DAYS = AnomalyRule::MIN_BASELINE_DAYS
16
+ # Volume metrics on a handful of events swing wildly; ignore them.
17
+ MIN_EVENTS = 20
18
+ COOL_DOWN = 60.minutes
19
+ MAX_GROUPS_PER_RULE = 50
20
+ MAX_CURRENT_ROWS_PER_RULE = MAX_GROUPS_PER_RULE * 25
21
+ MAX_BASELINE_ROWS_PER_RULE = 50_000
22
+
23
+ def perform(environment)
24
+ now = Time.current
25
+ environment.anomaly_rules.enabled.find_each do |rule|
26
+ detect(environment, rule, now).each { |detection| open_issue(environment, rule, now, detection) }
27
+ end
28
+ end
29
+
30
+ private
31
+
32
+ def detect(environment, rule, now)
33
+ from = now - rule.window_minutes.minutes
34
+ type = TYPE_FOR.fetch(rule.target_kind)
35
+ environment.with_telemetry do
36
+ groups = current_groups(rule, type, from, now)
37
+ baseline = baseline_rows(rule, type, groups.keys, from, now)
38
+ return [] unless baseline
39
+
40
+ groups.filter_map do |group_hash, (name, summary)|
41
+ anomaly_for(rule, group_hash, name, summary, baseline.fetch(group_hash, []), from, now)
42
+ end
43
+ end
44
+ end
45
+
46
+ # Rollups are hourly, so a window shorter than an hour reads the whole
47
+ # enclosing hour bucket (Rollup.between rounds `from` down); the baseline
48
+ # windows are shifted by whole days and land on the same hour, so the
49
+ # comparison stays like-for-like.
50
+ def current_groups(rule, type, from, to)
51
+ scope = Telemetry::Rollup.for_type(type).between(from, to)
52
+ scope = scope.where(name: rule.target) unless rule.target == "*"
53
+ group_hashes = scope.reorder(nil).group(:group_hash).order(Arel.sql("SUM(count) DESC"), :group_hash)
54
+ .limit(MAX_GROUPS_PER_RULE + 1).pluck(:group_hash)
55
+ if group_hashes.length > MAX_GROUPS_PER_RULE
56
+ Rails.logger.warn("anomaly detector evaluated only the #{MAX_GROUPS_PER_RULE} busiest groups rule_id=#{rule.id}")
57
+ group_hashes.pop
58
+ end
59
+ rows = scope.where(group_hash: group_hashes).order(:bucket, :group_hash).limit(MAX_CURRENT_ROWS_PER_RULE + 1).to_a
60
+ if rows.length > MAX_CURRENT_ROWS_PER_RULE
61
+ Rails.logger.error("anomaly detector skipped rule_id=#{rule.id}: more than #{MAX_CURRENT_ROWS_PER_RULE} current rollups")
62
+ return {}
63
+ end
64
+
65
+ rows.group_by(&:group_hash)
66
+ .transform_values { |group_rows| [ group_rows.first.name, Telemetry::Rollup.summarize(group_rows) ] }
67
+ end
68
+
69
+ def anomaly_for(rule, group_hash, name, summary, baseline_rows, from, to)
70
+ return if cooling_down?(rule, group_hash)
71
+ current = value_for(rule.metric, summary, rule.window_minutes)
72
+ return if current.nil?
73
+ return if %w[count error_rate].include?(rule.metric) && summary[:count] < MIN_EVENTS
74
+
75
+ baseline = baseline_values(rule, baseline_rows, from, to)
76
+ return if baseline.size < MIN_BASELINE_DAYS
77
+ mean = baseline.sum / baseline.size
78
+ stddev = Math.sqrt(baseline.sum { |v| (v - mean)**2 } / baseline.size)
79
+ return unless current > mean + (rule.deviation * stddev) && current > mean * 1.25
80
+
81
+ # A perfectly flat baseline has no σ to measure against, so report the
82
+ # rule's own threshold rather than an infinite one.
83
+ sigmas = stddev.positive? ? (current - mean) / stddev : rule.deviation
84
+ { group_hash: group_hash, name: name, current: current.round(2), mean: mean.round(2),
85
+ stddev: stddev.round(2), sigmas: sigmas.round(2) }
86
+ end
87
+
88
+ # The same clock window on each of the previous `baseline_days` days. Days
89
+ # with no rollups at all are left out rather than counted as zero.
90
+ def baseline_values(rule, rows, from, to)
91
+ (1..rule.baseline_days).filter_map do |days|
92
+ day_rows = rows.select { |row| row.bucket.between?((from - days.days).beginning_of_hour, to - days.days) }
93
+ next if day_rows.empty?
94
+ value_for(rule.metric, Telemetry::Rollup.summarize(day_rows), rule.window_minutes)
95
+ end
96
+ end
97
+
98
+ # One query for every group's baseline, reading only the hour buckets the
99
+ # baseline windows actually cover rather than the whole `baseline_days` span.
100
+ def baseline_rows(rule, type, group_hashes, from, to)
101
+ return {} if group_hashes.empty?
102
+
103
+ rows = Telemetry::Rollup.for_type(type)
104
+ .where(group_hash: group_hashes, bucket: baseline_buckets(rule, from, to))
105
+ .order(:bucket, :group_hash).limit(MAX_BASELINE_ROWS_PER_RULE + 1).to_a
106
+ if rows.length > MAX_BASELINE_ROWS_PER_RULE
107
+ Rails.logger.error("anomaly detector skipped rule_id=#{rule.id}: more than #{MAX_BASELINE_ROWS_PER_RULE} baseline rollups")
108
+ return nil
109
+ end
110
+ rows.group_by(&:group_hash)
111
+ end
112
+
113
+ # The hour buckets covered by the same clock window on each of the previous
114
+ # `baseline_days` days.
115
+ def baseline_buckets(rule, from, to)
116
+ (1..rule.baseline_days).flat_map do |days|
117
+ bucket = (from - days.days).beginning_of_hour
118
+ last = to - days.days
119
+ buckets = []
120
+ while bucket <= last
121
+ buckets << bucket
122
+ bucket += 1.hour
123
+ end
124
+ buckets
125
+ end
126
+ end
127
+
128
+ def value_for(metric, summary, window_minutes)
129
+ case metric
130
+ when "p95" then summary[:p95] / 1000.0
131
+ when "avg" then summary[:avg] / 1000.0
132
+ when "count" then summary[:count] / window_minutes.to_f
133
+ when "error_rate" then summary[:count].zero? ? nil : (summary[:errors] * 100.0 / summary[:count])
134
+ end
135
+ end
136
+
137
+ # Reads the primary database: one open anomaly issue per (rule, group) is
138
+ # enough for an hour, however often the scan runs.
139
+ def cooling_down?(rule, group_hash)
140
+ Issue.where(environment_id: rule.environment_id, group_hash: issue_group_hash(rule, group_hash))
141
+ .where(last_seen_at: COOL_DOWN.ago..).exists?
142
+ end
143
+
144
+ def issue_group_hash(rule, group_hash)
145
+ "anomaly:#{rule.id}:#{group_hash}"
146
+ end
147
+
148
+ def open_issue(environment, rule, now, detection)
149
+ issue, outcome = Issue.record_occurrence!(
150
+ environment: environment, group_hash: issue_group_hash(rule, detection[:group_hash]), kind: "anomaly",
151
+ title: "#{rule.metric} of #{detection[:name]} is #{detection[:sigmas].round(1)}σ above its #{rule.baseline_days}-day baseline (#{detection[:current]} vs #{detection[:mean]})",
152
+ culprit: detection[:name], occurred_at: now, deploy: nil, user_ref: nil,
153
+ sample: { rule_id: rule.id, metric: rule.metric, current: detection[:current], mean: detection[:mean],
154
+ stddev: detection[:stddev], sigmas: detection[:sigmas], window_minutes: rule.window_minutes })
155
+ carry_detection(issue, outcome) do
156
+ IssueDetectionSnapshot.capture(
157
+ environment: environment, kind: "anomaly", telemetry_type: TYPE_FOR.fetch(rule.target_kind),
158
+ telemetry_group_hash: detection[:group_hash], target: detection[:name], metric: rule.metric,
159
+ from: now - rule.window_minutes.minutes, to: now, value: detection[:current],
160
+ baseline: detection.slice(:mean, :stddev, :sigmas).merge(days: rule.baseline_days, deviation: rule.deviation),
161
+ rule: { type: "anomaly", id: rule.id, target_kind: rule.target_kind, target: rule.target }
162
+ )
163
+ end
164
+ if outcome == :new || outcome == :regressed
165
+ issue.fire_alerts!("anomaly", rule_id: rule.id, metric: rule.metric, current: detection[:current],
166
+ mean: detection[:mean], sigmas: detection[:sigmas])
167
+ end
168
+ rule.update_columns(last_fired_at: now)
169
+ end
170
+ end
171
+ end
@@ -0,0 +1,85 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Railwatch
4
+ # Evaluates every threshold for an environment against the last window of
5
+ # rollups and opens performance issues. Runs on a schedule (recurring.yml).
6
+ class DetectPerformanceIssuesJob < ApplicationJob
7
+ include DetectionSnapshotting
8
+
9
+ queue_as :default
10
+
11
+ MAX_GROUPS_PER_RULE = 200
12
+ MAX_ROLLUP_ROWS_PER_RULE = MAX_GROUPS_PER_RULE * 25
13
+
14
+ TYPE_FOR = { "requests" => "request", "jobs" => "job_attempt", "commands" => "command", "queries" => "query",
15
+ "scheduled_tasks" => "scheduled_task", "outgoing_requests" => "outgoing_request" }.freeze
16
+
17
+ def perform(environment)
18
+ now = Time.current
19
+ environment.thresholds.find_each do |threshold|
20
+ from = now - threshold.window_minutes.minutes
21
+ type = TYPE_FOR.fetch(threshold.target_kind)
22
+ groups = environment.with_telemetry do
23
+ scope = Telemetry::Rollup.for_type(type).between(from, now)
24
+ scope = scope.where(name: threshold.target) unless threshold.target == "*"
25
+ bounded_groups(scope, threshold)
26
+ end
27
+ groups.each do |group_hash, (name, summary)|
28
+ value = value_for(threshold.metric, summary)
29
+ next if value.nil? || value <= threshold.limit
30
+ open_issue(environment, threshold, group_hash, name, value, from, now)
31
+ end
32
+ end
33
+ end
34
+
35
+ private
36
+
37
+ def bounded_groups(scope, threshold)
38
+ group_hashes = scope.reorder(nil).group(:group_hash).order(Arel.sql("SUM(count) DESC"), :group_hash)
39
+ .limit(MAX_GROUPS_PER_RULE + 1).pluck(:group_hash)
40
+ if group_hashes.length > MAX_GROUPS_PER_RULE
41
+ Rails.logger.warn("threshold detector evaluated only the #{MAX_GROUPS_PER_RULE} busiest groups threshold_id=#{threshold.id}")
42
+ group_hashes.pop
43
+ end
44
+ rows = scope.where(group_hash: group_hashes).order(:bucket, :group_hash).limit(MAX_ROLLUP_ROWS_PER_RULE + 1).to_a
45
+ if rows.length > MAX_ROLLUP_ROWS_PER_RULE
46
+ Rails.logger.error("threshold detector skipped threshold_id=#{threshold.id}: more than #{MAX_ROLLUP_ROWS_PER_RULE} rollups")
47
+ return {}
48
+ end
49
+
50
+ rows.group_by(&:group_hash)
51
+ .transform_values { |group_rows| [ group_rows.first.name, Telemetry::Rollup.summarize(group_rows) ] }
52
+ end
53
+
54
+ def value_for(metric, s)
55
+ case metric
56
+ when "p95" then s[:p95] / 1000.0
57
+ when "max" then s[:max] / 1000.0
58
+ when "avg" then s[:avg] / 1000.0
59
+ when "error_rate", "failure_rate" then s[:count].zero? ? nil : (s[:errors] * 100.0 / s[:count])
60
+ end
61
+ end
62
+
63
+ def open_issue(environment, threshold, group_hash, name, value, from, now)
64
+ issue, outcome = Issue.record_occurrence!(
65
+ environment: environment, group_hash: "perf:#{threshold.id}:#{group_hash}", kind: "performance",
66
+ title: "#{name} exceeded #{threshold.metric} #{threshold.limit}#{threshold.metric.end_with?('rate') ? '%' : 'ms'} (#{value.round(1)})",
67
+ culprit: name, occurred_at: now, deploy: nil, user_ref: nil,
68
+ sample: { threshold_id: threshold.id, value: value.round(2), metric: threshold.metric, limit: threshold.limit })
69
+ carry_detection(issue, outcome) do
70
+ IssueDetectionSnapshot.capture(
71
+ environment: environment, kind: "performance", telemetry_type: TYPE_FOR.fetch(threshold.target_kind),
72
+ telemetry_group_hash: group_hash, target: name, metric: threshold.metric, from: from, to: now,
73
+ value: value.round(2), limit: threshold.limit,
74
+ rule: { type: "threshold", id: threshold.id, target_kind: threshold.target_kind, target: threshold.target }
75
+ )
76
+ end
77
+ return unless outcome == :new || outcome == :regressed
78
+ payload = { issue_key: issue.key, title: issue.title, environment: environment.name }
79
+ environment.application.alert_rules.where(event: "threshold").find_each do |rule|
80
+ next unless rule.matches?(issue: issue, payload: payload)
81
+ rule.fire!(event: "threshold", issue: issue, payload: payload)
82
+ end
83
+ end
84
+ end
85
+ end
@@ -0,0 +1,91 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Railwatch
4
+ # Turns newly ingested exception rows into Issues (create or bump), and
5
+ # fires new / regressed alerts.
6
+ class GroupExceptionsJob < ApplicationJob
7
+ queue_as :default
8
+
9
+ # A browser whose in-flight request died while kamal-proxy swapped
10
+ # containers reports a network error a few seconds after the deploy
11
+ # marker. That is the deploy working, not the app failing, and it opened
12
+ # (then regressed) an issue on every deploy today. The exception rows are
13
+ # still recorded; they just do not become an issue when they land inside
14
+ # this window around a deploy. Same idea as CheckScheduledTasksJob's
15
+ # DEPLOY_GRACE for missed runs.
16
+ DEPLOY_GRACE = 3.minutes
17
+ DEPLOY_NETWORK_ERRORS = %w[HttpNetworkError AxiosError InertiaException NetworkError TypeError:NetworkError].freeze
18
+
19
+ # batch_id: the embedded ledger's id for the batch these rows came from.
20
+ # With one, each group's occurrence count is committed together with a
21
+ # FollowupReceipt for (batch, group), so running this again for the same
22
+ # batch (a crash after the count but before the ledger was cleared, or
23
+ # the inline drain racing the maintenance drain) counts nothing twice.
24
+ # The hosted platform enqueues this once per HTTP batch and passes none.
25
+ def perform(environment, exception_ids, batch_id: nil)
26
+ rows = environment.with_telemetry { Telemetry::Exception.where(id: exception_ids).to_a }
27
+ rows = rows.reject { |row| deploy_swap_noise?(environment, row) }
28
+ rows.group_by(&:group_hash).each do |group_hash, group|
29
+ issue = outcome = nil
30
+ Issue.transaction do
31
+ next if batch_id && !FollowupReceipt.claim!(batch_id: batch_id, group_hash: group_hash)
32
+
33
+ latest = group.max_by(&:occurred_at)
34
+ issue, outcome = Issue.record_occurrence!(
35
+ environment: environment, group_hash: group_hash, kind: "exception",
36
+ title: "#{latest.class_name}: #{latest.message.to_s.first(200)}",
37
+ culprit: [ latest.file, latest.line ].compact.join(":").presence,
38
+ occurred_at: latest.occurred_at, deploy: latest.deploy, user_ref: latest.user_ref, source: latest.source,
39
+ sample: { exception_id: latest.id, handled: latest.handled, execution_id: latest.execution_id,
40
+ execution_preview: latest.execution_preview, deploy: latest.deploy,
41
+ fingerprint: latest.fingerprint, fingerprint_source: latest.fingerprint_source })
42
+ issue.increment!(:occurrences, group.size - 1) if group.size > 1
43
+ end
44
+ next unless issue
45
+
46
+ update_affected_users(environment, issue) if affected_users_due?(issue, outcome)
47
+ alert(issue, outcome)
48
+ end
49
+ end
50
+
51
+ # The affected-user count is a DISTINCT over every retained occurrence of
52
+ # the group, so on a busy issue it grows with the retention window and
53
+ # used to run once per batch per group. Once per window per issue, and
54
+ # always for a new issue, keeps the number fresh at a bounded cost.
55
+ AFFECTED_USERS_WINDOW = 5.minutes
56
+ AFFECTED_USERS_AT = Concurrent::Map.new
57
+
58
+ private
59
+
60
+ def affected_users_due?(issue, outcome)
61
+ now = Time.current
62
+ return AFFECTED_USERS_AT[issue.id] = now if outcome == :new
63
+
64
+ last = AFFECTED_USERS_AT[issue.id]
65
+ return false if last && now - last < AFFECTED_USERS_WINDOW
66
+
67
+ AFFECTED_USERS_AT[issue.id] = now
68
+ end
69
+
70
+ def deploy_swap_noise?(environment, row)
71
+ return false unless row.source == "browser" && DEPLOY_NETWORK_ERRORS.include?(row.class_name)
72
+
73
+ last_deploy_at = environment.deploys.maximum(:deployed_at) or return false
74
+ row.occurred_at.between?(last_deploy_at - DEPLOY_GRACE, last_deploy_at + DEPLOY_GRACE)
75
+ end
76
+
77
+ def update_affected_users(environment, issue)
78
+ count = environment.with_telemetry { Telemetry::Exception.where(group_hash: issue.group_hash).where.not(user_ref: nil).distinct.count(:user_ref) }
79
+ issue.update_columns(affected_users: count)
80
+ end
81
+
82
+ def alert(issue, outcome)
83
+ event = { new: "new_issue", regressed: "regressed_issue" }[outcome] or return
84
+ payload = { issue_key: issue.key, title: issue.title, environment: issue.environment.name }
85
+ issue.application.alert_rules.where(event: event).find_each do |rule|
86
+ next unless rule.matches?(issue: issue, payload: payload)
87
+ rule.fire!(event: event, issue: issue, payload: payload)
88
+ end
89
+ end
90
+ end
91
+ end
@@ -0,0 +1,23 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Railwatch
4
+ # Refreshes SQLite's planner statistics for each environment's telemetry
5
+ # database. Nothing else ever runs ANALYZE on these files, so the planner
6
+ # has been choosing index shapes blind on tables that grow by hundreds of
7
+ # thousands of rows a day. PRAGMA optimize only re-analyzes tables whose
8
+ # stats look stale, so it is cheap to run daily; analysis_limit bounds the
9
+ # rows it samples per index.
10
+ class OptimizeTelemetryJob < ApplicationJob
11
+ queue_as :maintenance
12
+
13
+ def perform(environment = nil)
14
+ return [ Environment.current ].each { |env| self.class.perform_later(env) } if environment.nil?
15
+
16
+ environment.with_telemetry do
17
+ connection = TelemetryRecord.connection
18
+ connection.execute("PRAGMA analysis_limit = 1000")
19
+ connection.execute("PRAGMA optimize")
20
+ end
21
+ end
22
+ end
23
+ end
@@ -0,0 +1,11 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Railwatch
4
+ class PerformanceScanJob < ApplicationJob
5
+ queue_as :default
6
+
7
+ def perform
8
+ [ Environment.current ].each { |env| DetectPerformanceIssuesJob.perform_later(env) }
9
+ end
10
+ end
11
+ end
@@ -0,0 +1,110 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Railwatch
4
+ # Deletes raw rows past the account's retention window. Rollups are kept
5
+ # for 13 months so long-range charts keep working after raw rows are gone.
6
+ # Telemetry::Session is monitored-application telemetry; dashboard login
7
+ # sessions live in the primary database and are deliberately not touched.
8
+ class PruneTelemetryJob < ApplicationJob
9
+ queue_as :maintenance
10
+
11
+ AGGREGATE_RETENTION = 13.months
12
+ # Sessions have never been pruned, so the first run of this job would walk
13
+ # every hour an environment has ever recorded. Delete a bounded number of
14
+ # hourly buckets per run and let the nightly schedule drain the backlog.
15
+ SESSION_HOURS_PER_RUN = 48
16
+ BATCH = 5_000
17
+ # Per raw table per run. A backlog past this (a retention change, a
18
+ # restore of an old file) drains over successive runs rather than in one
19
+ # that holds SQLite's write lock for as long as it takes. In the embedded
20
+ # writer that long hold would look like a wedge and end the process.
21
+ MAX_BATCHES_PER_TABLE = 40
22
+
23
+ RAW = [ Telemetry::Execution, Telemetry::Query, Telemetry::Exception, Telemetry::CacheEvent, Telemetry::Mail,
24
+ Telemetry::Broadcast, Telemetry::Notification, Telemetry::OutgoingRequest, Telemetry::StorageOp,
25
+ Telemetry::ViewRender, Telemetry::Log, Telemetry::EnqueuedJob, Telemetry::Transaction,
26
+ Telemetry::NPlusOne, Telemetry::Deprecation, Telemetry::Visit, Telemetry::Span,
27
+ Telemetry::LlmCall,
28
+ Telemetry::Profile, Telemetry::Attachment ].freeze
29
+
30
+ # checkpoint: the WAL checkpoint mode run at the end. TRUNCATE (the
31
+ # default, for a dedicated worker) hands the space back to the filesystem
32
+ # but blocks every reader and writer while it does; PASSIVE checkpoints
33
+ # what it can without waiting on anyone, which is what a prune running
34
+ # inside a Puma worker (Railwatch::Maintenance) must use.
35
+ def perform(environment = nil, checkpoint: "TRUNCATE")
36
+ return [ Environment.current ].each { |env| self.class.perform_later(env) } if environment.nil?
37
+
38
+ cutoff = environment.retention_days.days.ago
39
+ aggregate_cutoff = AGGREGATE_RETENTION.ago
40
+ environment.with_telemetry do
41
+ unindex_logs(cutoff) if Telemetry::Log.fts_available?
42
+ RAW.each { |klass| prune(klass, cutoff) }
43
+ # NOT EXISTS rather than NOT IN: queries.group_hash is nullable, and one
44
+ # NULL in a NOT IN subquery makes it match nothing. A few hundred shapes.
45
+ Telemetry::QueryShape.where("NOT EXISTS (SELECT 1 FROM queries WHERE queries.group_hash = query_shapes.group_hash)").delete_all
46
+ prune_sessions(cutoff)
47
+ Telemetry::Rollup.where(bucket: ...aggregate_cutoff).delete_all
48
+ Telemetry::ReleaseHealth.where(bucket: ...aggregate_cutoff).in_batches(of: 5_000).delete_all
49
+ Telemetry::Person.where(last_seen_at: ...cutoff).in_batches(of: 5_000).delete_all
50
+ Telemetry::IngestBatch.where(received_at: ...cutoff).delete_all
51
+ Telemetry::Process.where(booted_at: ...cutoff).delete_all
52
+ Telemetry::HealthSample.where(sampled_at: ...cutoff).delete_all
53
+ TelemetryRecord.connection.execute("PRAGMA wal_checkpoint(#{checkpoint})") if TelemetryRecord.connection.adapter_name =~ /sqlite/i
54
+ end
55
+ end
56
+
57
+ private
58
+
59
+ # in_batches walks a table by primary key: its probe is "WHERE occurred_at
60
+ # < ? ORDER BY id LIMIT n", which no index serves, so SQLite scanned the
61
+ # whole table by rowid to find nothing to delete -- 42 seconds on the
62
+ # platform's own 12M-row queries table, every night, for an empty result.
63
+ # Ordered by occurred_at the same probe is one seek on any index led by
64
+ # that column, and the first page of expired rows is exactly the oldest
65
+ # ones. Tables without such an index still scan, but they are the small
66
+ # ones.
67
+ def prune(klass, cutoff)
68
+ MAX_BATCHES_PER_TABLE.times do
69
+ # Ordered by (occurred_at, id), not occurred_at alone: rows that tie
70
+ # on the timestamp at the limit boundary would otherwise be a
71
+ # different set here than in unindex_logs above, which leaves stale
72
+ # FTS postings behind and withdraws postings for logs that are staying.
73
+ ids = klass.where(occurred_at: ...cutoff).order(:occurred_at, :id).limit(BATCH).pluck(:id)
74
+ break if ids.empty?
75
+ klass.where(id: ids).delete_all
76
+ break if ids.size < BATCH
77
+ end
78
+ end
79
+
80
+ # Sessions are the raw input to the hourly release_health aggregate, so they
81
+ # are deleted a whole hour at a time. Flooring the cutoff to the start of its
82
+ # hour leaves the partial hour straddling the boundary in place: deleting
83
+ # only its expired prefix would let the next rollup rebuild that hour from
84
+ # the surviving tail and silently shrink an already correct aggregate.
85
+ def prune_sessions(cutoff)
86
+ expired = Telemetry::Session.where(occurred_at: ...cutoff.utc.beginning_of_hour)
87
+ SESSION_HOURS_PER_RUN.times do
88
+ oldest = expired.minimum(:occurred_at)
89
+ break if oldest.nil?
90
+
91
+ hour = oldest.utc.beginning_of_hour
92
+ expired.where(occurred_at: hour...(hour + 1.hour)).in_batches(of: 5_000).delete_all
93
+ end
94
+ end
95
+
96
+ # logs_fts is an external-content index with no triggers, so the postings
97
+ # for a log line have to be withdrawn while its row (and message) is still
98
+ # there. Deleting the rows first would leave the index permanently out of
99
+ # sync -- searches would keep returning rowids that no longer exist.
100
+ # Bounded the same way as the rows it precedes: the postings for at most
101
+ # MAX_BATCHES_PER_TABLE * BATCH expired lines are withdrawn per run, which
102
+ # is exactly the set prune(Telemetry::Log) will delete this run.
103
+ def unindex_logs(cutoff)
104
+ Telemetry::Log.connection.execute(Telemetry::Log.sanitize_sql_array([
105
+ "INSERT INTO logs_fts(logs_fts, rowid, message) SELECT 'delete', id, message FROM logs " \
106
+ "WHERE occurred_at < ? ORDER BY occurred_at, id LIMIT ?", cutoff, MAX_BATCHES_PER_TABLE * BATCH
107
+ ]))
108
+ end
109
+ end
110
+ end
@@ -0,0 +1,64 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Railwatch
4
+ # Recomputes one hourly release_health bucket from raw `sessions` rows.
5
+ # Idempotent, and debounced by the same Solid Queue concurrency control
6
+ # RollupJob uses: while one recompute of a bucket is queued or running,
7
+ # further enqueues for it are discarded rather than piling up.
8
+ #
9
+ # The gem re-sends a live session every flush interval, so the same
10
+ # session_id lands in a bucket many times. Each id collapses to the worst
11
+ # status it reached (crashed beats errored beats ok beats started) and its
12
+ # longest reported duration, which is what makes "sessions" a session count
13
+ # rather than a record count.
14
+ class ReleaseHealthRollupJob < ApplicationJob
15
+ queue_as :rollups
16
+ limits_concurrency to: 1, key: ->(environment, bucket) { "release_health:#{environment.id}:#{bucket.to_i}" }, duration: 10.minutes, on_conflict: :discard
17
+
18
+ RANKS = { "started" => 0, "ok" => 1, "errored" => 2, "crashed" => 3 }.freeze
19
+ CRASHED = RANKS["crashed"]
20
+ ERRORED = RANKS["errored"]
21
+
22
+ def perform(environment, bucket)
23
+ bucket = bucket.utc.beginning_of_hour
24
+ # PruneTelemetryJob deletes raw sessions in whole hours below this floor.
25
+ # A rollup enqueued before the prune can run after it, and rebuilding a
26
+ # pruned hour would replace a complete aggregate with an empty one.
27
+ return if bucket < environment.retention_days.days.ago.utc.beginning_of_hour
28
+
29
+ environment.with_telemetry do
30
+ rows = Telemetry::Session.where(occurred_at: bucket...(bucket + 1.hour))
31
+ .where.not(deploy: nil).pluck(:deploy, :session_id, :status, :duration, :user_ref)
32
+ aggregates = rows.group_by(&:first).map { |deploy, group| aggregate(deploy, bucket, collapse(group)) }
33
+ Telemetry::ReleaseHealth.transaction do
34
+ Telemetry::ReleaseHealth.where(bucket: bucket).delete_all
35
+ Telemetry::ReleaseHealth.insert_all(aggregates) if aggregates.any?
36
+ end
37
+ end
38
+ end
39
+
40
+ private
41
+
42
+ # session_id => the worst status, longest duration, and first user seen for
43
+ # it in this bucket.
44
+ def collapse(rows)
45
+ rows.each_with_object({}) do |(_deploy, session_id, status, duration, user_ref), sessions|
46
+ session = sessions[session_id] ||= { rank: 0, duration: nil, user: nil }
47
+ session[:rank] = [ session[:rank], RANKS.fetch(status.to_s, 0) ].max
48
+ session[:duration] = [ session[:duration] || 0, duration ].max if duration
49
+ session[:user] ||= user_ref
50
+ end
51
+ end
52
+
53
+ def aggregate(deploy, bucket, sessions)
54
+ durations = sessions.values.filter_map { |s| s[:duration] }
55
+ users = sessions.values.filter_map { |s| s[:user] }.uniq
56
+ crashed_users = sessions.values.select { |s| s[:rank] == CRASHED }.filter_map { |s| s[:user] }.uniq
57
+ { deploy: deploy, bucket: bucket, sessions: sessions.size,
58
+ sessions_errored: sessions.values.count { |s| s[:rank] == ERRORED },
59
+ sessions_crashed: sessions.values.count { |s| s[:rank] == CRASHED },
60
+ users: users.size, users_crashed: crashed_users.size,
61
+ duration_sum: durations.sum, duration_count: durations.size }
62
+ end
63
+ end
64
+ end
@@ -0,0 +1,21 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Railwatch
4
+ # Recomputes the current and previous hour so rollups are never more than a
5
+ # schedule tick stale, whatever happened to the per-batch RollupJob enqueues
6
+ # (debounced in the web process, and their concurrency semaphore can outlive
7
+ # a worker restart). The platform iterates every active environment; an
8
+ # embedded install has one.
9
+ class RollupCatchupJob < ApplicationJob
10
+ queue_as :rollups
11
+
12
+ def perform
13
+ env = Environment.current
14
+ now = Time.current
15
+ [ now.beginning_of_hour, (now - 1.hour).beginning_of_hour ].each do |bucket|
16
+ RollupJob.perform_now(env, bucket)
17
+ ReleaseHealthRollupJob.perform_now(env, bucket)
18
+ end
19
+ end
20
+ end
21
+ end