railwatch 0.3.4 → 0.3.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (167) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +32 -0
  3. data/app/jobs/railwatch/prune_telemetry_job.rb +20 -1
  4. data/app/models/railwatch/telemetry_record.rb +48 -0
  5. data/db/railwatch_telemetry_migrate/20260903000000_enable_incremental_vacuum.rb +62 -0
  6. data/docs/embedded.md +47 -1
  7. data/lib/railwatch/version.rb +1 -1
  8. data/lib/tasks/railwatch_tasks.rake +161 -0
  9. data/public/railwatch/assets/{app-layout-yh-sPWgK.js → app-layout-CdJ5Qjms.js} +1 -1
  10. data/public/railwatch/assets/{app-wordmark-nbjkzxwQ.js → app-wordmark-B-0yZiDb.js} +1 -1
  11. data/public/railwatch/assets/{appearance-CDuRvQTB.js → appearance-CU28GBf1.js} +1 -1
  12. data/public/railwatch/assets/application-CwshqwM6.css +1 -0
  13. data/public/railwatch/assets/{arrow-up-DVtOGdVA.js → arrow-up-DY1fS0tD.js} +1 -1
  14. data/public/railwatch/assets/{auth-layout-TCwPpS1K.js → auth-layout-C3FKuUJz.js} +1 -1
  15. data/public/railwatch/assets/{badge-Daw4Hvr8.js → badge-DKvItdlb.js} +1 -1
  16. data/public/railwatch/assets/{braces-DhFbHPsz.js → braces-Bse5Gh3K.js} +1 -1
  17. data/public/railwatch/assets/{card-BX_3HXcJ.js → card-C9j_ffR9.js} +1 -1
  18. data/public/railwatch/assets/{chart-B14-N9g7.js → chart-Cmh6JEww.js} +1 -1
  19. data/public/railwatch/assets/{chart-hover-Vy52H4uD.js → chart-hover-B0SjyWFy.js} +1 -1
  20. data/public/railwatch/assets/{chart-panel-Cb4S_ej_.js → chart-panel-DKNxoauI.js} +1 -1
  21. data/public/railwatch/assets/{checkbox-D3aRSsBj.js → checkbox-BsBkRg5n.js} +1 -1
  22. data/public/railwatch/assets/{code-KvW8k7Jr.js → code-C879Qwgg.js} +1 -1
  23. data/public/railwatch/assets/{copy-block-BKFGd11J.js → copy-block-DtO_3esq.js} +1 -1
  24. data/public/railwatch/assets/{copy-id-Djks1fXB.js → copy-id-SeLCKsW2.js} +1 -1
  25. data/public/railwatch/assets/{cursor-load-more-BXV0f1D_.js → cursor-load-more-OpHbe_kE.js} +1 -1
  26. data/public/railwatch/assets/{data-table-iv1bdF6u.js → data-table-Bc3ZRnpb.js} +1 -1
  27. data/public/railwatch/assets/{edit-__9yJlO3.js → edit-BMpgcQhU.js} +1 -1
  28. data/public/railwatch/assets/{edit-DMKUF8Zi.js → edit-CTD1jSWh.js} +1 -1
  29. data/public/railwatch/assets/{edit-BZc_Iawe.js → edit-DDvf9mkq.js} +1 -1
  30. data/public/railwatch/assets/{empty-state-6j_0AaQQ.js → empty-state-CkPROZB0.js} +1 -1
  31. data/public/railwatch/assets/env-layout-aGQgTZRg.js +1 -0
  32. data/public/railwatch/assets/{execution-path-CzgBUi5e.js → execution-path-BX5NcXUZ.js} +1 -1
  33. data/public/railwatch/assets/{filter-bar-9SU5NrzX.js → filter-bar-C0O2rJVr.js} +1 -1
  34. data/public/railwatch/assets/{flamegraph-qcekju8V.js → flamegraph-Vj4QSAlT.js} +1 -1
  35. data/public/railwatch/assets/{frames-Cyu7KMxZ.js → frames-CTiGwm4E.js} +1 -1
  36. data/public/railwatch/assets/{google-sign-in-button-DsTSfmzY.js → google-sign-in-button-BzogBGeF.js} +1 -1
  37. data/public/railwatch/assets/index-B-kuWp-m.js +1 -0
  38. data/public/railwatch/assets/{index-1ol1-QWI.js → index-B2_QluAw.js} +1 -1
  39. data/public/railwatch/assets/index-B69Iad2U.js +1 -0
  40. data/public/railwatch/assets/index-BCLvKVhP.js +1 -0
  41. data/public/railwatch/assets/index-BKgAm-w2.js +1 -0
  42. data/public/railwatch/assets/index-BRN_bXV_.js +1 -0
  43. data/public/railwatch/assets/index-Ble4RoA3.js +1 -0
  44. data/public/railwatch/assets/index-Bvy021Vv.js +1 -0
  45. data/public/railwatch/assets/index-CCLMaY9k.js +1 -0
  46. data/public/railwatch/assets/index-CD4HIHq0.js +1 -0
  47. data/public/railwatch/assets/{index-Cmlwy1-V.js → index-CPQKOQxV.js} +1 -1
  48. data/public/railwatch/assets/{index-CeYKnapu.js → index-CY6lbwAC.js} +1 -1
  49. data/public/railwatch/assets/index-C__S10ze.js +1 -0
  50. data/public/railwatch/assets/index-ChQ1_u8u.js +1 -0
  51. data/public/railwatch/assets/index-CtT4T6kp.js +1 -0
  52. data/public/railwatch/assets/index-CxU1Jj35.js +1 -0
  53. data/public/railwatch/assets/index-D3Axxkk2.js +1 -0
  54. data/public/railwatch/assets/index-D5YVcrqR.js +1 -0
  55. data/public/railwatch/assets/index-D9cV-IQr.js +1 -0
  56. data/public/railwatch/assets/index-DbyRgbzQ.js +1 -0
  57. data/public/railwatch/assets/index-Dcxd_H-U.js +1 -0
  58. data/public/railwatch/assets/index-Dep2XomH.js +1 -0
  59. data/public/railwatch/assets/{index-DlumCsWY.js → index-DepurF6s.js} +2 -2
  60. data/public/railwatch/assets/{index-BRiq3SNR.js → index-DigUOHQ8.js} +1 -1
  61. data/public/railwatch/assets/index-DkgImCOg.js +1 -0
  62. data/public/railwatch/assets/{index-DU9F5b5d.js → index-DxuDLoTc.js} +1 -1
  63. data/public/railwatch/assets/{index-OqI59zPb.js → index-Dy7Wg880.js} +1 -1
  64. data/public/railwatch/assets/index-FgpFDA1Z.js +1 -0
  65. data/public/railwatch/assets/index-LaTKDe0-.js +1 -0
  66. data/public/railwatch/assets/index-Wb_U2_hz.js +1 -0
  67. data/public/railwatch/assets/index-chS28dvS.js +1 -0
  68. data/public/railwatch/assets/index-jSEU_Yp7.js +1 -0
  69. data/public/railwatch/assets/index-uuKg6Zqn.js +1 -0
  70. data/public/railwatch/assets/index-y2c8xKzI.js +1 -0
  71. data/public/railwatch/assets/index-z2YfKBJl.js +1 -0
  72. data/public/railwatch/assets/{inertia-TViv6kNv.js → inertia-CgBdJDaE.js} +2 -2
  73. data/public/railwatch/assets/{input-error-LxImUkxv.js → input-error-DfpIawjl.js} +1 -1
  74. data/public/railwatch/assets/{json-viewer-Ar4cjPDW.js → json-viewer-CH7nEIUs.js} +1 -1
  75. data/public/railwatch/assets/klass-BTg1l2AO.js +1 -0
  76. data/public/railwatch/assets/{label-COUKWqE_.js → label-BWgQzHCO.js} +1 -1
  77. data/public/railwatch/assets/{layout-DNSLAkw_.js → layout-neknbxn-.js} +1 -1
  78. data/public/railwatch/assets/{live-dot-ChfUtY3p.js → live-dot-Bejk0a5o.js} +1 -1
  79. data/public/railwatch/assets/{nav-CNnDqPlm.js → nav-C1M8941-.js} +1 -1
  80. data/public/railwatch/assets/{new-D4uewYC8.js → new-B8CzyoMu.js} +1 -1
  81. data/public/railwatch/assets/{new-Be55nmt9.js → new-BnGIYM37.js} +1 -1
  82. data/public/railwatch/assets/{new-Bi_xQiIb.js → new-DElC-JCS.js} +1 -1
  83. data/public/railwatch/assets/{new-D05SajFR.js → new-DGYPcz7w.js} +1 -1
  84. data/public/railwatch/assets/{new-84S8ZJq9.js → new-DLgCLBiW.js} +1 -1
  85. data/public/railwatch/assets/{new-BvCT8TMg.js → new-bBwawsEg.js} +1 -1
  86. data/public/railwatch/assets/{onboarding-CYZi5Cqc.js → onboarding-CW0jUhHT.js} +1 -1
  87. data/public/railwatch/assets/{origin-identity-6q1-CBts.js → origin-identity-bJMxeEHB.js} +1 -1
  88. data/public/railwatch/assets/{percentile-picker-gFZCXtdb.js → percentile-picker-DLAnTKY3.js} +1 -1
  89. data/public/railwatch/assets/{relative-time-IOOgl5n2.js → relative-time-BHbJh8HW.js} +1 -1
  90. data/public/railwatch/assets/release-health-2O1GRclB.js +1 -0
  91. data/public/railwatch/assets/route-DSw8aIZa.js +1 -0
  92. data/public/railwatch/assets/{segmented-h1VdDTqE.js → segmented-Dg8CES94.js} +1 -1
  93. data/public/railwatch/assets/{select-_AJsUa7X.js → select-CqKpvI98.js} +1 -1
  94. data/public/railwatch/assets/{separator-BwwTYtCF.js → separator-BOp2LHw3.js} +1 -1
  95. data/public/railwatch/assets/{series-chart-DaFPefku.js → series-chart-COEtj3PY.js} +1 -1
  96. data/public/railwatch/assets/show-B40560ZT.js +1 -0
  97. data/public/railwatch/assets/{show-C4Ltf5i9.js → show-B7Zqjlqc.js} +1 -1
  98. data/public/railwatch/assets/{show-CeTL4B37.js → show-BAxMoNvw.js} +2 -2
  99. data/public/railwatch/assets/show-B_QVoqKE.js +1 -0
  100. data/public/railwatch/assets/show-Bbs1i46N.js +1 -0
  101. data/public/railwatch/assets/{show-DACku6AD.js → show-C-xpDkoj.js} +3 -3
  102. data/public/railwatch/assets/{show-DgSIoRvA.js → show-C9Y1To9_.js} +1 -1
  103. data/public/railwatch/assets/show-CGv85sRo.js +1 -0
  104. data/public/railwatch/assets/{show-BM6X2Mpo.js → show-CPh-7VGL.js} +1 -1
  105. data/public/railwatch/assets/show-C_9M2nLy.js +1 -0
  106. data/public/railwatch/assets/{show-DIOSGcXV.js → show-DFlP1ktM.js} +2 -2
  107. data/public/railwatch/assets/show-DqtE0LI6.js +1 -0
  108. data/public/railwatch/assets/{show-C8sHalnw.js → show-DrhEDGPn.js} +1 -1
  109. data/public/railwatch/assets/{show-JxFtB4eK.js → show-HjjzcKRq.js} +2 -2
  110. data/public/railwatch/assets/show-Zr2xV4OD.js +1 -0
  111. data/public/railwatch/assets/show-dRDz8OVu.js +1 -0
  112. data/public/railwatch/assets/{show-DYteoYWW.js → show-pbkvGRER.js} +1 -1
  113. data/public/railwatch/assets/{sort-header-DpFzXblu.js → sort-header-DBxQ_KFX.js} +1 -1
  114. data/public/railwatch/assets/{sparkline-cell-C3-5vFkP.js → sparkline-cell-BbfXw3kA.js} +1 -1
  115. data/public/railwatch/assets/stat-Cd8QaTGw.js +1 -0
  116. data/public/railwatch/assets/{status-badge-8jVV-LA4.js → status-badge-Ys4b0fcG.js} +1 -1
  117. data/public/railwatch/assets/{tenant-path-G-6u9A-o.js → tenant-path-CQWzaNYR.js} +1 -1
  118. data/public/railwatch/assets/{text-link-DfsiaCcP.js → text-link-BY8FY-CJ.js} +1 -1
  119. data/public/railwatch/assets/{textarea-Dye72uP7.js → textarea-Bq2jjOMC.js} +1 -1
  120. data/public/railwatch/assets/{timeline-CD7WHnbo.js → timeline-SxStCFO2.js} +1 -1
  121. data/public/railwatch/assets/{transition-B_AW8rMK.js → transition-CQ9yvfPa.js} +1 -1
  122. data/public/railwatch/assets/{use-clipboard-ColgLyQ2.js → use-clipboard-3hTXU91g.js} +1 -1
  123. data/public/railwatch/assets/use-live-LwJStJvD.js +1 -0
  124. data/public/railwatch/manifest.json +1289 -1236
  125. metadata +117 -115
  126. data/public/railwatch/assets/application-DD4nJA5I.css +0 -1
  127. data/public/railwatch/assets/env-layout-DrT8rO6P.js +0 -1
  128. data/public/railwatch/assets/index-5jI4aFzC.js +0 -1
  129. data/public/railwatch/assets/index-9KTrVnrc.js +0 -1
  130. data/public/railwatch/assets/index-B0-8lcTp.js +0 -1
  131. data/public/railwatch/assets/index-B7jjfNfO.js +0 -1
  132. data/public/railwatch/assets/index-BBchRy0M.js +0 -1
  133. data/public/railwatch/assets/index-BgKj9xhr.js +0 -1
  134. data/public/railwatch/assets/index-BkTZqqOu.js +0 -1
  135. data/public/railwatch/assets/index-BprKx8QO.js +0 -1
  136. data/public/railwatch/assets/index-C3jzvPs3.js +0 -1
  137. data/public/railwatch/assets/index-Cbs6gGyQ.js +0 -1
  138. data/public/railwatch/assets/index-CdRZ6AWF.js +0 -1
  139. data/public/railwatch/assets/index-Cwx6058d.js +0 -1
  140. data/public/railwatch/assets/index-D4CSdbHv.js +0 -1
  141. data/public/railwatch/assets/index-DEFMSkdG.js +0 -1
  142. data/public/railwatch/assets/index-DFiHEBSh.js +0 -1
  143. data/public/railwatch/assets/index-DL4vWdWJ.js +0 -1
  144. data/public/railwatch/assets/index-DaXgPcGL.js +0 -1
  145. data/public/railwatch/assets/index-DbtaU-EE.js +0 -1
  146. data/public/railwatch/assets/index-DeOe83F4.js +0 -1
  147. data/public/railwatch/assets/index-DiucHN4B.js +0 -1
  148. data/public/railwatch/assets/index-DlnR_l9o.js +0 -1
  149. data/public/railwatch/assets/index-DmRd7aIG.js +0 -1
  150. data/public/railwatch/assets/index-DxSh2UpM.js +0 -1
  151. data/public/railwatch/assets/index-MIMGuFNt.js +0 -1
  152. data/public/railwatch/assets/index-P4rC7IlX.js +0 -1
  153. data/public/railwatch/assets/index-gpPOcFWq.js +0 -1
  154. data/public/railwatch/assets/index-oVkururr.js +0 -1
  155. data/public/railwatch/assets/index-p9puqVge.js +0 -1
  156. data/public/railwatch/assets/klass-CJ-J4INB.js +0 -1
  157. data/public/railwatch/assets/release-health-DC8oc7uw.js +0 -1
  158. data/public/railwatch/assets/route-Dv6LAWvT.js +0 -1
  159. data/public/railwatch/assets/show-B7NCgkEo.js +0 -1
  160. data/public/railwatch/assets/show-BKqyKjBK.js +0 -1
  161. data/public/railwatch/assets/show-BNw4tN5q.js +0 -1
  162. data/public/railwatch/assets/show-BO3bnG5h.js +0 -1
  163. data/public/railwatch/assets/show-BhrAVAEA.js +0 -1
  164. data/public/railwatch/assets/show-CpfgV1jP.js +0 -1
  165. data/public/railwatch/assets/show-DQp_1n-B.js +0 -1
  166. data/public/railwatch/assets/show-DVNz46RI.js +0 -1
  167. data/public/railwatch/assets/stat-s4RpOS9w.js +0 -1
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: 3d212beaade3dc390933ea8a41bd49eb1bbf2c896056071516aac8242b39eb07
4
- data.tar.gz: e647976f9634c1e0ba390f6dec52866d190401728363e3dcd1fa93aa2976af3d
3
+ metadata.gz: 8904a8f775771bd894a52f80b7d5ecf4d8a40d0dee09aaf16697e68685c5e106
4
+ data.tar.gz: de27dad2204510d9e8f3d1bb3ada89901ccd467486c94fde9f0b85ff83ab50ca
5
5
  SHA512:
6
- metadata.gz: c669c17f558dec72be16f175b06abc5769680591349c2169f6598fc26575191558bd00b903ead0f1f4287bab718c5097b0550ff60e7cfe5fe82fa53940c7a953
7
- data.tar.gz: fc9a32a877b22e12e22b7738b4b479c2aa8b9d1dcc5c97f7ed003f5b421ea3e1a95f917221c195c9b01a487ac124d413dc561a22f96e7483e4891bb5db95b685
6
+ metadata.gz: 5fe5b808c60a3aac92302f8669b82c539634cb655d7317e98279143c5e920f9178d15920a1e75145c9c153ad5d3bfc41ade5004c1df9f1e7c8de29ab2fc20b76
7
+ data.tar.gz: 251237b9c8ee2a3d252c61eef492d89a532e3c7fef856bde91e19d4d3baa5c831ebc34a0bbbd24b2ced390f76f0ef8106c43cee2d68d6fbf25725f01b074298f
data/CHANGELOG.md CHANGED
@@ -1,5 +1,37 @@
1
1
  # Changelog
2
2
 
3
+ ## 0.3.5 (2026-09-20)
4
+
5
+ - Make the embedded telemetry database give disk back. `PruneTelemetryJob`
6
+ deleted rows past `retention_days` and nothing ever vacuumed, so the freed
7
+ pages went on SQLite's freelist to be reused and never returned to the
8
+ filesystem: the file only ever grew. Measured on a copy of a real production
9
+ database of the same shape (bulk deletes, continuous inserts), deleting
10
+ 40,148 rows left the file at 21M with 5,164 pages on the freelist; a single
11
+ `PRAGMA incremental_vacuum` took it to 44K.
12
+
13
+ Two halves, and neither is worth anything alone. A new telemetry migration,
14
+ `EnableIncrementalVacuum`, puts the database into `auto_vacuum=incremental`;
15
+ and the nightly prune now runs `PRAGMA incremental_vacuum` after its deletes,
16
+ bounded to 2,000 pages a slice and 25 slices (~200MB at SQLite's 4K default)
17
+ so a backlog drains over successive nights instead of stalling one.
18
+
19
+ The mode cannot be declared in `config/database.yml`: Rails applies its own
20
+ `DEFAULT_PRAGMAS` before any you declare, and `journal_mode = wal` writes the
21
+ file header, so by the time `auto_vacuum` runs the database is no longer new
22
+ and SQLite accepts the statement and ignores it. The migration is numbered
23
+ below `CreateTelemetry` for the same reason -- it has to run while the file
24
+ still holds nothing.
25
+
26
+ - Add `bin/rails railwatch:vacuum:status` and `bin/rails railwatch:vacuum`.
27
+ An **existing** install cannot change mode without a full `VACUUM`, which
28
+ rewrites the whole file with the write lock held, so nothing does that on its
29
+ own: the migration leaves an existing database exactly as it found it.
30
+ `railwatch:vacuum:status` reports the file, its size, its mode and its
31
+ freelist and changes nothing; `railwatch:vacuum` says what the conversion
32
+ will cost and then does it. Installs created from this version on need
33
+ neither.
34
+
3
35
  ## 0.3.4 (2026-09-20)
4
36
 
5
37
  - Fix the embedded dashboard offering its install steps to an install that is
@@ -20,6 +20,21 @@ module Railwatch
20
20
  # writer that long hold would look like a wedge and end the process.
21
21
  MAX_BATCHES_PER_TABLE = 40
22
22
 
23
+ # Reclaim, bounded the same way the deletes are. Deleting rows only moves
24
+ # their pages to the freelist; PRAGMA incremental_vacuum is what hands
25
+ # them back to the filesystem, and it is only possible at all on a
26
+ # database in auto_vacuum=incremental (EnableIncrementalVacuum, or
27
+ # `bin/rails railwatch:vacuum` for a database that predates it).
28
+ # Measured at 80k-190k pages/s warm, so a full run is well under a second
29
+ # there; the slicing is for the cold, large file, where each slice takes
30
+ # and releases the write lock instead of holding it throughout.
31
+ VACUUM_PAGES_PER_SLICE = 2_000
32
+ # 50k pages, ~200MB at SQLite's 4KB default. A few thousand pages a night
33
+ # would never keep up with a night's deletes on a busy app, and the file
34
+ # would go on growing with the freelist; a backlog past this one still
35
+ # drains over successive nightly runs.
36
+ VACUUM_SLICES = 25
37
+
23
38
  RAW = [ Telemetry::Execution, Telemetry::Query, Telemetry::Exception, Telemetry::CacheEvent, Telemetry::Mail,
24
39
  Telemetry::Broadcast, Telemetry::Notification, Telemetry::OutgoingRequest, Telemetry::StorageOp,
25
40
  Telemetry::ViewRender, Telemetry::Log, Telemetry::EnqueuedJob, Telemetry::Transaction,
@@ -50,7 +65,11 @@ module Railwatch
50
65
  Telemetry::IngestBatch.where(received_at: ...cutoff).delete_all
51
66
  Telemetry::Process.where(booted_at: ...cutoff).delete_all
52
67
  Telemetry::HealthSample.where(sampled_at: ...cutoff).delete_all
53
- TelemetryRecord.connection.execute("PRAGMA wal_checkpoint(#{checkpoint})") if TelemetryRecord.connection.adapter_name =~ /sqlite/i
68
+ # Before the checkpoint, not after: incremental_vacuum truncates the
69
+ # database file, and in WAL mode that truncation only reaches the file
70
+ # on disk once it is checkpointed.
71
+ TelemetryRecord.reclaim_freelist!(slice: VACUUM_PAGES_PER_SLICE, slices: VACUUM_SLICES)
72
+ TelemetryRecord.connection.execute("PRAGMA wal_checkpoint(#{checkpoint})") if TelemetryRecord.sqlite?
54
73
  end
55
74
  end
56
75
 
@@ -31,6 +31,54 @@ module Railwatch
31
31
  end
32
32
  end
33
33
 
34
+ # SQLite's auto_vacuum modes. Deleted pages only leave the file in
35
+ # :incremental, and only when PRAGMA incremental_vacuum asks for them;
36
+ # :none (the default, and what every database created before
37
+ # EnableIncrementalVacuum is in) keeps them on the freelist forever.
38
+ AUTO_VACUUM_MODES = { 0 => :none, 1 => :full, 2 => :incremental }.freeze
39
+
40
+ # Everything below reads and writes THIS database, never the host's. The
41
+ # pragmas are per-database and there is no Active Record wrapper for them,
42
+ # so they go through this class's own connection on purpose: through
43
+ # ActiveRecord::Base they would report on, and vacuum, the application's
44
+ # primary database instead.
45
+ def self.sqlite? = connection.adapter_name.match?(/sqlite/i)
46
+
47
+ def self.auto_vacuum_mode = sqlite? ? AUTO_VACUUM_MODES.fetch(connection.select_value("PRAGMA auto_vacuum").to_i, :unknown) : nil
48
+
49
+ def self.freelist_pages = sqlite? ? connection.select_value("PRAGMA freelist_count").to_i : 0
50
+
51
+ def self.page_size = sqlite? ? connection.select_value("PRAGMA page_size").to_i : 0
52
+
53
+ # Hands freelist pages back to the filesystem, a slice at a time. Each
54
+ # PRAGMA is its own implicit transaction, so the write lock is taken and
55
+ # released once per slice rather than held for the whole reclaim -- the
56
+ # same reason PruneTelemetryJob deletes in bounded batches instead of one
57
+ # statement. Returns the pages actually reclaimed, which is 0 on a
58
+ # database that is not in incremental mode: there the pragma is accepted
59
+ # and does nothing, and calling it would look like work that happened.
60
+ def self.reclaim_freelist!(slice:, slices:)
61
+ return 0 unless sqlite? && auto_vacuum_mode == :incremental
62
+
63
+ before = freelist_pages
64
+ slices.times do
65
+ break if freelist_pages.zero?
66
+
67
+ incremental_vacuum(slice.to_i)
68
+ end
69
+ before - freelist_pages
70
+ end
71
+
72
+ # Through the raw connection on purpose. incremental_vacuum frees one page
73
+ # per sqlite3_step, and Active Record's execute steps a statement with no
74
+ # result columns exactly once -- so through it this pragma frees a single
75
+ # page whatever count it is handed, which is the kind of no-op that reads
76
+ # as work done. The sqlite3 gem's own execute steps to completion. Same
77
+ # raw_connection route Ingest::Writer already takes for its inserts.
78
+ def self.incremental_vacuum(pages = nil)
79
+ connection.raw_connection.execute("PRAGMA incremental_vacuum#{"(#{pages})" if pages}")
80
+ end
81
+
34
82
  # Records arrive as the gem's wire hashes; this is the shared envelope.
35
83
  def self.envelope_columns(t)
36
84
  t.datetime :occurred_at, null: false, precision: 6
@@ -0,0 +1,62 @@
1
+ # frozen_string_literal: true
2
+
3
+ # Puts the telemetry database into incremental auto-vacuum mode, which is what
4
+ # lets PruneTelemetryJob hand the pages it frees back to the filesystem
5
+ # instead of parking them on the freelist forever. Without it the file only
6
+ # ever grows: deleting a day of telemetry frees pages for SQLite to reuse and
7
+ # returns not one byte to the disk.
8
+ #
9
+ # Why a migration, and why this version number. SQLite only lets a database
10
+ # leave auto_vacuum=none for free while it still holds no pages; after that
11
+ # the only way in is a full VACUUM, which rewrites the entire file with the
12
+ # write lock held. Declaring the pragma in config/database.yml cannot do it
13
+ # either: Rails applies its own DEFAULT_PRAGMAS before any declared ones, and
14
+ # `journal_mode = wal` writes the file header, so by the time
15
+ # `auto_vacuum = incremental` runs the database is no longer new and the
16
+ # statement is a silent no-op (measured: the mode comes out `none`).
17
+ #
18
+ # A migration is the one hook that runs both on a new install's db:prepare and
19
+ # on an existing install's upgrade, and this one is numbered below
20
+ # CreateTelemetry so that on a new database it runs while the file holds
21
+ # nothing but schema_migrations and ar_internal_metadata -- a few kilobytes,
22
+ # where the VACUUM that commits the mode is instant. Rails runs a pending
23
+ # migration whatever its version, so an existing install gets it too; there it
24
+ # does nothing at all, because converting a file that may be tens of gigabytes
25
+ # is not a decision a deploy gets to make. That one is `bin/rails
26
+ # railwatch:vacuum`, run deliberately, which says what it will cost first.
27
+ class EnableIncrementalVacuum < ActiveRecord::Migration[8.1]
28
+ # VACUUM cannot run inside a transaction.
29
+ disable_ddl_transaction!
30
+
31
+ # Rails' own bookkeeping, which is present before the first migration runs
32
+ # and so does not make a database "existing".
33
+ BOOKKEEPING = %w[schema_migrations ar_internal_metadata].freeze
34
+
35
+ def up
36
+ return say("not SQLite (#{connection.adapter_name}); auto_vacuum does not apply") unless sqlite?
37
+ return say("auto_vacuum is already incremental") if mode == "incremental"
38
+
39
+ populated = connection.tables - BOOKKEEPING
40
+ if populated.any?
41
+ return say("#{populated.size} tables already here: left in auto_vacuum=#{mode}. Run " \
42
+ "`bin/rails railwatch:vacuum` to convert this database when you can spare the lock.")
43
+ end
44
+
45
+ connection.execute("PRAGMA auto_vacuum = incremental")
46
+ connection.execute("VACUUM")
47
+ say("auto_vacuum = #{mode}")
48
+ end
49
+
50
+ # Irreversible in the useful sense: going back to `none` is another whole
51
+ # VACUUM, and nothing about the schema depends on the mode. Rolling back
52
+ # leaves it where it is rather than spending that on an undo nobody asked
53
+ # for.
54
+ def down
55
+ say("auto_vacuum left as #{sqlite? ? mode : connection.adapter_name}; use `PRAGMA auto_vacuum = none; VACUUM;` to undo it by hand")
56
+ end
57
+
58
+ private
59
+ def sqlite? = connection.adapter_name.match?(/sqlite/i)
60
+
61
+ def mode = { 0 => "none", 1 => "full", 2 => "incremental" }[connection.select_value("PRAGMA auto_vacuum").to_i]
62
+ end
data/docs/embedded.md CHANGED
@@ -322,7 +322,7 @@ The work that keeps the dashboard current happens in two places:
322
322
  | anomaly scan | every 5 minutes | Anomaly rules, when any are enabled |
323
323
  | scheduled tasks | every 10 minutes | Missed and late scheduled tasks |
324
324
  | auto-resolve | daily | Resolves issues quiet for 14 days |
325
- | prune | daily | Deletes telemetry older than `retention_days`, then `ANALYZE` |
325
+ | prune | daily | Deletes telemetry older than `retention_days`, returns the freed pages to the filesystem when incremental auto-vacuum is on (see below), then `ANALYZE` |
326
326
 
327
327
  Because the clock lives in the web process, it keeps running when the
328
328
  job worker is down, which is exactly when "scheduled task X missed its
@@ -375,6 +375,52 @@ telemetry database grows with traffic and sampling and is pruned to
375
375
  SQLite files in `storage/`, so Litestream or a volume snapshot covers
376
376
  them.
377
377
 
378
+ ### Giving the disk back
379
+
380
+ Deleting rows does not shrink a SQLite file on its own. The pages go on
381
+ the database's freelist, where later inserts reuse them, and the file
382
+ stays whatever size it reached. So pruning alone keeps the *contents*
383
+ bounded and lets the *file* grow forever.
384
+
385
+ Railwatch creates its telemetry database in SQLite's
386
+ `auto_vacuum=incremental` mode, and the nightly prune runs `PRAGMA
387
+ incremental_vacuum` after its deletes, which hands those pages back to
388
+ the filesystem. It is bounded -- 2,000 pages a slice, 25 slices, about
389
+ 200MB a night at SQLite's 4K default page size -- so a large backlog
390
+ drains over successive nights rather than holding the write lock through
391
+ one enormous reclaim. Nothing to configure.
392
+
393
+ ```
394
+ $ bin/rails railwatch:vacuum:status
395
+ Railwatch telemetry database
396
+ file /rails/storage/production_railwatch_telemetry.sqlite3
397
+ size 1.42 GB
398
+ mode auto_vacuum=incremental
399
+ free 312 pages (1.22 MB) on the freelist
400
+
401
+ The nightly prune returns up to 195 MB a night on its own.
402
+ `bin/rails railwatch:vacuum` returns all 1.22 MB now.
403
+ ```
404
+
405
+ **Installs created before 0.3.5 report `auto_vacuum=none`, and pruning
406
+ cannot return their space.** SQLite can only set the mode on a database
407
+ that is still empty; on one that has data the only route is a full
408
+ `VACUUM`, which rewrites the entire file with the write lock held.
409
+ Railwatch will not do that to a running application behind your back, so
410
+ upgrading leaves the mode alone and the conversion is a task you run
411
+ when you can spare the lock:
412
+
413
+ ```
414
+ $ bin/rails railwatch:vacuum
415
+ ```
416
+
417
+ It prints the file, its size and its freelist first, then says how long
418
+ the `VACUUM` should take and that it needs about the file's own size in
419
+ free disk for the temporary copy. Telemetry written while it runs waits
420
+ for it and the dashboard is frozen for the duration, so pick a quiet
421
+ moment. Once converted, the nightly prune keeps up on its own and you
422
+ never need to run it again.
423
+
378
424
  ## Switching to the cloud later
379
425
 
380
426
  Set a token and drop `c.transport = :local` (or set
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Railwatch
4
- VERSION = "0.3.4"
4
+ VERSION = "0.3.5"
5
5
  end
@@ -28,6 +28,130 @@ module Railwatch
28
28
  end
29
29
  end
30
30
 
31
+ # What the embedded telemetry database is costing on disk, and the one-time
32
+ # conversion that lets pruning give that cost back.
33
+ #
34
+ # Deleting rows does not shrink a SQLite file. The pages go on the freelist
35
+ # and are reused by later inserts, and the only way back to the filesystem
36
+ # is auto_vacuum=incremental plus PRAGMA incremental_vacuum -- which
37
+ # PruneTelemetryJob now runs after every prune. A database can only enter
38
+ # that mode while it is still empty (EnableIncrementalVacuum does that for
39
+ # every telemetry database created since it shipped) or through a full
40
+ # VACUUM, which is what this offers to an older one.
41
+ class TelemetryDisk
42
+ # Deliberately pessimistic, and said out loud as an estimate: VACUUM
43
+ # rewrites the whole file, and what that costs is the operator's disk,
44
+ # not ours to know.
45
+ VACUUM_BYTES_PER_SECOND = 50 * 1024 * 1024
46
+
47
+ # nil, having said why, for an install with no telemetry database to
48
+ # talk about.
49
+ def self.open
50
+ unless Railwatch.config.local?
51
+ puts "Railwatch reports over HTTP here (transport = :http), so this app has no telemetry database; nothing to vacuum."
52
+ return nil
53
+ end
54
+
55
+ environment = Railwatch::Environment.current
56
+ return new(environment) if environment.with_telemetry { Railwatch::TelemetryRecord.sqlite? }
57
+
58
+ puts "The railwatch_telemetry database is not SQLite; auto_vacuum does not apply."
59
+ nil
60
+ end
61
+
62
+ def initialize(environment)
63
+ @environment = environment
64
+ end
65
+
66
+ def report
67
+ with_telemetry do
68
+ <<~TEXT.chomp
69
+ Railwatch telemetry database
70
+ file #{path}
71
+ size #{human(file_bytes)}#{" + #{human(wal_bytes)} WAL" if wal_bytes.positive?}
72
+ mode auto_vacuum=#{mode}
73
+ free #{freelist} pages (#{human(freelist * page_size)}) on the freelist
74
+ TEXT
75
+ end
76
+ end
77
+
78
+ def advice
79
+ with_telemetry do
80
+ if mode != :incremental
81
+ "\nPruning cannot return space to the filesystem in this mode: every page it frees stays in this file.\n" \
82
+ "`bin/rails railwatch:vacuum` converts the database to incremental auto-vacuum with a full VACUUM. " \
83
+ "That rewrites all #{human(file_bytes)}, needs about that much free disk for the temporary copy, and " \
84
+ "holds the write lock for roughly #{estimate}. Telemetry written while it runs waits for it."
85
+ elsif freelist.positive?
86
+ "\nThe nightly prune returns up to #{human(Railwatch::PruneTelemetryJob::VACUUM_PAGES_PER_SLICE * Railwatch::PruneTelemetryJob::VACUUM_SLICES * page_size)} " \
87
+ "a night on its own. `bin/rails railwatch:vacuum` returns all #{human(freelist * page_size)} now."
88
+ else
89
+ "\nNothing on the freelist: pruning is already returning this database's space as it goes."
90
+ end
91
+ end
92
+ end
93
+
94
+ # The whole point of asking explicitly, so this one is not bounded the way
95
+ # the nightly prune is.
96
+ def reclaim!
97
+ with_telemetry do
98
+ before = disk_bytes
99
+ if mode == :incremental
100
+ Railwatch::TelemetryRecord.incremental_vacuum
101
+ else
102
+ puts "\nConverting to incremental auto-vacuum (full VACUUM, roughly #{estimate})..."
103
+ connection.execute("PRAGMA auto_vacuum = incremental")
104
+ connection.execute("VACUUM")
105
+ end
106
+ # A checkpoint that cannot finish says so in its result rather than
107
+ # raising: a reader still on an older snapshot holds the WAL open.
108
+ # Counting only the main file would then report the WAL's bytes as
109
+ # returned while they are still on the disk.
110
+ busy = checkpoint_busy?
111
+ [
112
+ "\nauto_vacuum=#{mode}, #{human(disk_bytes)} on disk including the WAL " \
113
+ "(#{human(before - disk_bytes)} returned), #{freelist} pages left on the freelist.",
114
+ busy ? "The WAL could not be truncated yet -- something is still reading it. " \
115
+ "Its bytes come back at the next checkpoint; run this again if you want to watch it." : nil
116
+ ].compact.join("\n")
117
+ end
118
+ end
119
+
120
+ private
121
+ def with_telemetry(&) = @environment.with_telemetry(&)
122
+
123
+ def connection = Railwatch::TelemetryRecord.connection
124
+
125
+ def mode = Railwatch::TelemetryRecord.auto_vacuum_mode
126
+
127
+ def freelist = Railwatch::TelemetryRecord.freelist_pages
128
+
129
+ def page_size = Railwatch::TelemetryRecord.page_size
130
+
131
+ def path = Rails.root.join(Railwatch::TelemetryRecord.connection_db_config.database.to_s)
132
+
133
+ def file_bytes = File.exist?(path) ? File.size(path) : 0
134
+
135
+ # What this database actually occupies. In WAL mode the pages a VACUUM
136
+ # frees are not off the disk until the WAL is checkpointed, so the main
137
+ # file alone understates it -- and, right after a vacuum, flatters it.
138
+ def disk_bytes = file_bytes + wal_bytes
139
+
140
+ # `PRAGMA wal_checkpoint` answers with [busy, log_pages, checkpointed];
141
+ # a non-zero first column means it gave up rather than failed.
142
+ def checkpoint_busy?
143
+ connection.select_rows("PRAGMA wal_checkpoint(TRUNCATE)").dig(0, 0).to_i != 0
144
+ rescue StandardError
145
+ false
146
+ end
147
+
148
+ def wal_bytes = File.exist?("#{path}-wal") ? File.size("#{path}-wal") : 0
149
+
150
+ def estimate = ActiveSupport::Duration.build([ (file_bytes / VACUUM_BYTES_PER_SECOND.to_f).ceil, 1 ].max).inspect
151
+
152
+ def human(bytes) = ActiveSupport::NumberHelper.number_to_human_size(bytes)
153
+ end
154
+
31
155
  # Where this install's platform lives, derived from config.ingest_url:
32
156
  # railwatch:token and railwatch:mcp point at the same host the gem already
33
157
  # ships to, so a self-hosted app never gets told to visit railwatch.rebulk.com.
@@ -127,6 +251,25 @@ namespace :railwatch do
127
251
  else "cannot check: #{pending}"
128
252
  end, fatal: true)
129
253
  end
254
+ # An install created before 0.3.5 is in auto_vacuum=none, where the
255
+ # nightly prune's reclaim is a silent no-op and the file only grows.
256
+ # Nothing converts it on its own, because that needs a full VACUUM with
257
+ # the write lock held -- so the check people actually run is where it
258
+ # has to be said, rather than leaving them to notice the disk.
259
+ if Railwatch.config.local? && Railwatch::TelemetryRecord.sqlite?
260
+ vacuum_mode = begin
261
+ Railwatch::Environment.current.with_telemetry { Railwatch::TelemetryRecord.auto_vacuum_mode }
262
+ rescue StandardError => e
263
+ e.message
264
+ end
265
+ check.call(vacuum_mode == :incremental, "telemetry disk",
266
+ case vacuum_mode
267
+ when :incremental then "auto_vacuum=incremental; the nightly prune returns freed pages"
268
+ when Symbol then "auto_vacuum=#{vacuum_mode}: pruned pages stay in the file " \
269
+ "(bin/rails railwatch:vacuum:status)"
270
+ else "cannot check: #{vacuum_mode}"
271
+ end)
272
+ end
130
273
  # The maintenance clock runs in web and worker processes, not in this
131
274
  # rake process, so what can be checked here is whether one has ticked.
132
275
  last_tick = begin
@@ -297,6 +440,24 @@ namespace :railwatch do
297
440
  puts "\nRailwatch is wired up."
298
441
  end
299
442
 
443
+ # Disk. `railwatch:vacuum:status` only reads; `railwatch:vacuum` is the
444
+ # one-time conversion a database created before EnableIncrementalVacuum
445
+ # needs, and it is a rake task rather than anything automatic because it
446
+ # rewrites the whole file with the write lock held.
447
+ desc "Report the telemetry database's size, free pages, and whether pruning can return them to the filesystem"
448
+ task "vacuum:status" => :environment do
449
+ disk = Railwatch::TelemetryDisk.open or next
450
+ puts disk.report
451
+ puts disk.advice
452
+ end
453
+
454
+ desc "Reclaim disk from the telemetry database. Converts it to incremental auto-vacuum if needed -- a full VACUUM, which locks the file"
455
+ task vacuum: :environment do
456
+ disk = Railwatch::TelemetryDisk.open or next
457
+ puts disk.report
458
+ puts disk.reclaim!
459
+ end
460
+
300
461
  namespace :export do
301
462
  desc "Show what the export queue is holding and whether it can send"
302
463
  task status: :environment do