railwatch 0.5.1 → 0.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (154) hide show
  1. checksums.yaml +4 -4
  2. data/AGENTS.md +45 -25
  3. data/CHANGELOG.md +147 -0
  4. data/README.md +50 -31
  5. data/app/controllers/railwatch/dashboard_controller.rb +1 -0
  6. data/app/jobs/railwatch/rollup_job.rb +3 -1
  7. data/app/models/railwatch/application_record.rb +2 -2
  8. data/app/models/railwatch/ingest/rollup_absorber.rb +1 -1
  9. data/app/models/railwatch/telemetry_record.rb +2 -2
  10. data/docs/ai-and-mcp.md +9 -3
  11. data/docs/configuration.md +97 -40
  12. data/docs/embedded.md +96 -43
  13. data/docs/faq.md +28 -15
  14. data/docs/getting-started.md +121 -42
  15. data/docs/records.md +13 -10
  16. data/docs/replacing-nightwatch.md +16 -14
  17. data/docs/replacing-sentry.md +19 -10
  18. data/docs/security.md +24 -3
  19. data/docs/self-hosting.md +9 -1
  20. data/docs/testing.md +14 -4
  21. data/docs/troubleshooting.md +72 -29
  22. data/lib/generators/railwatch/install/install_generator.rb +32 -16
  23. data/lib/generators/railwatch/install/templates/initializer.rb.tt +6 -6
  24. data/lib/puma/plugin/railwatch.rb +48 -3
  25. data/lib/railwatch/configuration.rb +16 -1
  26. data/lib/railwatch/engine.rb +14 -3
  27. data/lib/railwatch/reporter.rb +52 -16
  28. data/lib/railwatch/transport/http.rb +4 -9
  29. data/lib/railwatch/version.rb +1 -1
  30. data/lib/railwatch.rb +23 -1
  31. data/lib/tasks/railwatch_tasks.rake +11 -4
  32. data/llms.txt +21 -14
  33. data/public/railwatch/assets/{app-layout-DDyQa72H.js → app-layout-Zc0v-hYh.js} +1 -1
  34. data/public/railwatch/assets/{app-wordmark-o9CODKP0.js → app-wordmark-BA_60AVb.js} +1 -1
  35. data/public/railwatch/assets/{appearance-BwuCXabr.js → appearance-CcfP9tZ7.js} +1 -1
  36. data/public/railwatch/assets/application-BrN3Sz94.css +1 -0
  37. data/public/railwatch/assets/{arrow-up-C6PxDiY3.js → arrow-up-CLQ-7heQ.js} +1 -1
  38. data/public/railwatch/assets/{auth-layout-BRt8MGFD.js → auth-layout-C05gEBIQ.js} +1 -1
  39. data/public/railwatch/assets/{badge-CAxXV8za.js → badge-DRae8XwK.js} +1 -1
  40. data/public/railwatch/assets/{braces-DgomTCNf.js → braces-rSYydpLY.js} +1 -1
  41. data/public/railwatch/assets/{card-cAtqCxWl.js → card-DPjFKfen.js} +1 -1
  42. data/public/railwatch/assets/{chart-BBeBkkNa.js → chart-DWh7l8yM.js} +1 -1
  43. data/public/railwatch/assets/{chart-hover-B1M9jc0y.js → chart-hover-CfoZUY4J.js} +1 -1
  44. data/public/railwatch/assets/{chart-panel-DUQTz_C8.js → chart-panel-CC49WTQL.js} +1 -1
  45. data/public/railwatch/assets/{checkbox-CmhMHWZO.js → checkbox-DPkLUiwM.js} +1 -1
  46. data/public/railwatch/assets/{code-DESvxyTj.js → code-CLmYS6FU.js} +1 -1
  47. data/public/railwatch/assets/{copy-block-BkSU5832.js → copy-block-CGcXxp8J.js} +1 -1
  48. data/public/railwatch/assets/{copy-id-D03GhN9F.js → copy-id-vBYHQwxg.js} +1 -1
  49. data/public/railwatch/assets/{cursor-load-more-CRyuMeQb.js → cursor-load-more-Ddwxe5dZ.js} +1 -1
  50. data/public/railwatch/assets/{data-table-BIlt7Rtm.js → data-table-CzKTEE-O.js} +1 -1
  51. data/public/railwatch/assets/{edit-O0NSBWxo.js → edit-B1kmWkzd.js} +1 -1
  52. data/public/railwatch/assets/{edit-DJ0D0wHN.js → edit-D9cx4pbG.js} +1 -1
  53. data/public/railwatch/assets/{edit-Bb6MKoe4.js → edit-yv6j9p-V.js} +1 -1
  54. data/public/railwatch/assets/{empty-state-C38il627.js → empty-state-CW4wclK_.js} +1 -1
  55. data/public/railwatch/assets/{env-layout-REF7OM4q.js → env-layout-Kz7wks1x.js} +1 -1
  56. data/public/railwatch/assets/{execution-path-FYLq1TwC.js → execution-path-FAuIzOXB.js} +1 -1
  57. data/public/railwatch/assets/{filter-bar-CYog9Alp.js → filter-bar-CJDjWFib.js} +1 -1
  58. data/public/railwatch/assets/{flamegraph-DSs69foN.js → flamegraph-EaGkP2NT.js} +1 -1
  59. data/public/railwatch/assets/{frames-BUi2J5Mk.js → frames-zZuIaNH9.js} +1 -1
  60. data/public/railwatch/assets/{google-sign-in-button-BQiIKFdd.js → google-sign-in-button-2_zgbgVy.js} +1 -1
  61. data/public/railwatch/assets/{index-ZOGOB8SA.js → index-B49SWz7K.js} +1 -1
  62. data/public/railwatch/assets/{index-DqTFTP8p.js → index-BHlY4wKe.js} +1 -1
  63. data/public/railwatch/assets/{index-DrcKVG2f.js → index-BZpPtyFY.js} +1 -1
  64. data/public/railwatch/assets/{index-Dh4IRLFI.js → index-BeVK6jCL.js} +1 -1
  65. data/public/railwatch/assets/{index-umIAl-pL.js → index-BfbSo01U.js} +1 -1
  66. data/public/railwatch/assets/{index-BoUBioBP.js → index-BfgncAv6.js} +1 -1
  67. data/public/railwatch/assets/{index-C3A_9imx.js → index-BhNszK1k.js} +1 -1
  68. data/public/railwatch/assets/{index-tpz-OGUP.js → index-Bu01uWvw.js} +1 -1
  69. data/public/railwatch/assets/{index-CGs4m_fa.js → index-C1s_hK3p.js} +1 -1
  70. data/public/railwatch/assets/{index-ZSZg9rtq.js → index-C8Cggnbw.js} +1 -1
  71. data/public/railwatch/assets/{index-so4lRrRq.js → index-CBip6V4z.js} +1 -1
  72. data/public/railwatch/assets/{index-DvjY3dPD.js → index-CMJGss5R.js} +1 -1
  73. data/public/railwatch/assets/{index-DtHmuB9Q.js → index-CRo3yK20.js} +1 -1
  74. data/public/railwatch/assets/{index-r0tSIplE.js → index-CWB_p2J8.js} +1 -1
  75. data/public/railwatch/assets/{index-CICUIFHL.js → index-CWHpndGc.js} +1 -1
  76. data/public/railwatch/assets/{index-CFFpnzIS.js → index-Ca_S4Sc3.js} +1 -1
  77. data/public/railwatch/assets/{index-DSvlZVWG.js → index-CanPDDOa.js} +1 -1
  78. data/public/railwatch/assets/{index-CrZ3vHDL.js → index-Cie90Yat.js} +1 -1
  79. data/public/railwatch/assets/{index-FhUaPPab.js → index-CiepQ_pR.js} +1 -1
  80. data/public/railwatch/assets/{index-C_upSl_k.js → index-Cm1uCIGN.js} +1 -1
  81. data/public/railwatch/assets/{index-sTYvcbkh.js → index-CzitnnSC.js} +1 -1
  82. data/public/railwatch/assets/{index-C7OtLq_3.js → index-DEUfClv3.js} +1 -1
  83. data/public/railwatch/assets/index-DJKwo-mI.js +1 -0
  84. data/public/railwatch/assets/{index-DDI_Zx5V.js → index-DK6y0YHp.js} +1 -1
  85. data/public/railwatch/assets/{index-C-PmdhXA.js → index-DMNPpLH9.js} +1 -1
  86. data/public/railwatch/assets/{index-BaR1U9An.js → index-DZO1mSfX.js} +1 -1
  87. data/public/railwatch/assets/{index-CsoN51vW.js → index-Db5wj3M5.js} +1 -1
  88. data/public/railwatch/assets/{index-BiiyMcA0.js → index-Dcy5WktB.js} +1 -1
  89. data/public/railwatch/assets/{index-8-hnAhOD.js → index-DvG0-7Lx.js} +1 -1
  90. data/public/railwatch/assets/{index-QpTtwFwu.js → index-Dvj1wuka.js} +1 -1
  91. data/public/railwatch/assets/{index-DW2CBbxU.js → index-KyZX46qX.js} +1 -1
  92. data/public/railwatch/assets/{index-BeOh2t_S.js → index-Ze-KP-sl.js} +1 -1
  93. data/public/railwatch/assets/{index-CiPo4Gob.js → index-mhRbWLBM.js} +1 -1
  94. data/public/railwatch/assets/{index-CpkI015n.js → index-x099JL5f.js} +1 -1
  95. data/public/railwatch/assets/{index-JdCVBrw8.js → index-x28zb_nf.js} +1 -1
  96. data/public/railwatch/assets/{inertia-DLew8ZNx.js → inertia-Cuyz2ZHO.js} +2 -2
  97. data/public/railwatch/assets/{input-error-cvM6_Jht.js → input-error-hog6gGxg.js} +1 -1
  98. data/public/railwatch/assets/{json-viewer-D922McGi.js → json-viewer-DH2W9HXf.js} +1 -1
  99. data/public/railwatch/assets/{klass-CrwICqN8.js → klass-DHbDelLk.js} +1 -1
  100. data/public/railwatch/assets/{label-GWl7I6sf.js → label-DdCBgiUn.js} +1 -1
  101. data/public/railwatch/assets/{layout-0ZAnD3zl.js → layout-Cueyl7c5.js} +1 -1
  102. data/public/railwatch/assets/{live-dot-D1n_BreY.js → live-dot-BZgYTYdt.js} +1 -1
  103. data/public/railwatch/assets/{nav-DPxr1NNC.js → nav-BSSGObDZ.js} +1 -1
  104. data/public/railwatch/assets/{new-D-ZzUK9a.js → new-B8FSb8Bl.js} +1 -1
  105. data/public/railwatch/assets/{new-DEVkYv-z.js → new-C39_v2Ll.js} +1 -1
  106. data/public/railwatch/assets/{new-Cdl6pqST.js → new-DVOPwaC5.js} +1 -1
  107. data/public/railwatch/assets/{new-DHAHDrN7.js → new-Dd40nvJR.js} +1 -1
  108. data/public/railwatch/assets/{new-Dz4lZf1L.js → new-SxnYe1SE.js} +1 -1
  109. data/public/railwatch/assets/{new-GMrRFurX.js → new-eP5vKD3Y.js} +1 -1
  110. data/public/railwatch/assets/onboarding-CUpZl5KB.js +1 -0
  111. data/public/railwatch/assets/{origin-identity-Bk9yHWZ1.js → origin-identity-BYt2uiuo.js} +1 -1
  112. data/public/railwatch/assets/{percentile-picker-DfSx9yJO.js → percentile-picker-CeQgllxD.js} +1 -1
  113. data/public/railwatch/assets/{relative-time-CjIjb8Lg.js → relative-time-D5UbF4oO.js} +1 -1
  114. data/public/railwatch/assets/{release-health-4b3tivEf.js → release-health-uP-GF3_V.js} +1 -1
  115. data/public/railwatch/assets/{route-C_5BUtHK.js → route-DQAY8JEr.js} +1 -1
  116. data/public/railwatch/assets/{segmented-BgbT3wZa.js → segmented-BeIe4uqk.js} +1 -1
  117. data/public/railwatch/assets/{select-DmunxCKE.js → select-2R16457Z.js} +1 -1
  118. data/public/railwatch/assets/{separator-BXzEdZ_8.js → separator-D5I0UCB5.js} +1 -1
  119. data/public/railwatch/assets/series-chart-gnrzhmM6.js +1 -0
  120. data/public/railwatch/assets/show-2BkeNRUC.js +2 -0
  121. data/public/railwatch/assets/{show-mU38uGTg.js → show-B0X1hRQH.js} +1 -1
  122. data/public/railwatch/assets/{show-CAl7xcex.js → show-BKUV5l5q.js} +1 -1
  123. data/public/railwatch/assets/{show-DnR1Dnjd.js → show-BQJD_CsF.js} +1 -1
  124. data/public/railwatch/assets/{show-Dn-GwFZL.js → show-BhmD6Sxp.js} +1 -1
  125. data/public/railwatch/assets/{show-Y74rM0VT.js → show-C3KcnVo2.js} +1 -1
  126. data/public/railwatch/assets/{show-Dily73Xk.js → show-C95cHd06.js} +1 -1
  127. data/public/railwatch/assets/{show-DXs4deaC.js → show-CdB4uVQS.js} +1 -1
  128. data/public/railwatch/assets/{show-DI8IhNUH.js → show-CoCqcVIp.js} +1 -1
  129. data/public/railwatch/assets/{show-vQ4bndYD.js → show-D2LqjEsU.js} +1 -1
  130. data/public/railwatch/assets/{show-DcpTFiLi.js → show-DNpnKyTh.js} +1 -1
  131. data/public/railwatch/assets/{show-B2zLAW83.js → show-DOlTxbig.js} +1 -1
  132. data/public/railwatch/assets/{show-SvLOcPrx.js → show-DQCkttL8.js} +1 -1
  133. data/public/railwatch/assets/{show-BLpWUHWD.js → show-DcxesipC.js} +1 -1
  134. data/public/railwatch/assets/{show-DYskfl3-.js → show-IGJoc_X0.js} +1 -1
  135. data/public/railwatch/assets/{show-DSP9Cq_C.js → show-YsNpqbcI.js} +1 -1
  136. data/public/railwatch/assets/{show-DlRVS18-.js → show-qNV6H8SH.js} +1 -1
  137. data/public/railwatch/assets/{sort-header-Dcq9bzmo.js → sort-header-mEJUM3Av.js} +1 -1
  138. data/public/railwatch/assets/{sparkline-cell-BON3qQUB.js → sparkline-cell-Dwg7awwh.js} +1 -1
  139. data/public/railwatch/assets/{stat-DFEyFxkO.js → stat-ZtxGU8lE.js} +1 -1
  140. data/public/railwatch/assets/{status-badge-BaUKP7Yo.js → status-badge-CH5P-Xkj.js} +1 -1
  141. data/public/railwatch/assets/{tenant-path-DPZPc985.js → tenant-path-CQoP-BeF.js} +1 -1
  142. data/public/railwatch/assets/{text-link-BO77t9Xk.js → text-link-DHJ8BXx5.js} +1 -1
  143. data/public/railwatch/assets/{textarea-DTqrCiV0.js → textarea-BeQtQyl5.js} +1 -1
  144. data/public/railwatch/assets/{timeline-D5rJ0es2.js → timeline-CaKQVu48.js} +1 -1
  145. data/public/railwatch/assets/{transition-DMIrZVth.js → transition-ksDpqhKJ.js} +1 -1
  146. data/public/railwatch/assets/{use-clipboard-ByoUGQqA.js → use-clipboard-DNefo-ky.js} +1 -1
  147. data/public/railwatch/assets/{use-live-D7xKz2ma.js → use-live-DMTuhKfB.js} +1 -1
  148. data/public/railwatch/manifest.json +1286 -1286
  149. metadata +116 -116
  150. data/public/railwatch/assets/application-B7h1MIhi.css +0 -1
  151. data/public/railwatch/assets/index-CFRLPs4J.js +0 -1
  152. data/public/railwatch/assets/onboarding-D1vwaHYT.js +0 -1
  153. data/public/railwatch/assets/series-chart-Xf49v9cv.js +0 -1
  154. data/public/railwatch/assets/show-SHwZjXb7.js +0 -2
@@ -8,20 +8,48 @@ below are keyed to those lines.
8
8
  bin/rails railwatch:doctor
9
9
  ```
10
10
 
11
- The task exits non-zero only when **token** or **ingest reachable**
12
- fails. Everything else is informational. A `✗` there means a feature
11
+ The first lines depend on the transport. An embedded install checks its
12
+ databases, writer and dashboard; a cloud install checks its token and
13
+ ingest host. The task exits non-zero only on the lines marked fatal
14
+ below. Everything else is informational: a `✗` there means a feature
13
15
  isn't wired, not that the install is broken.
14
16
 
17
+ Embedded (`transport = :local`):
18
+
19
+ | Doctor line | What a `✗` means |
20
+ |---|---|
21
+ | `export` | Only shown with `export_enabled` on. Export is on and cannot work: no token, no URL, a plain-HTTP URL, or an unsupported policy. Fatal. |
22
+ | `export destination` | The export queue is blocked or deferred; the line gives the reason. `bin/rails railwatch:export:rebind` clears a credential block. |
23
+ | `railwatch database`, `railwatch_telemetry database` | The database is missing from `config/database.yml` for this environment. Fatal. Re-run the install generator. |
24
+ | `railwatch migrations`, `railwatch_telemetry migrations` | Migrations are pending. Fatal. Run `bin/rails db:prepare`. |
25
+ | `telemetry disk` | The telemetry database is not in incremental auto-vacuum, so pruning never shrinks the file. See [Embedded mode](embedded.md#giving-the-disk-back). |
26
+ | `maintenance` | No maintenance tick in the last ten minutes. Expected when the app is stopped: the clock runs in the app's processes, not in rake. |
27
+ | `writer process` | The writer socket is not answering, `plugin :railwatch` is missing from `config/puma.rb`, or the socket path is over Linux's 108-byte limit. Expected when the app is stopped. |
28
+ | `last write` | Shown when the writer answers: nothing written in five minutes. With the app serving traffic, the writer is stuck or workers are not reaching it. |
29
+ | `dashboard access` | HTTP Basic is on with no credentials outside development, so every page is 401; or Basic is off and nothing else is declared. Run `bin/rails railwatch:authentication:configure`, or see [Embedded mode](embedded.md#authentication). |
30
+ | `json compatibility` | The installed `json` gem cannot decode on this Rails; see [below](#binjobs-dies-in-a-loop-with-wrong-number-of-arguments-given-2-expected-1). |
31
+ | `recurring.yml` | `config/recurring.yml` still lists `Railwatch::*` jobs from a pre-release. Remove them. |
32
+
33
+ Cloud (`transport = :http`):
34
+
15
35
  | Doctor line | What a `✗` means |
16
36
  |---|---|
17
37
  | `token` | `RAILWATCH_TOKEN` is unset or empty. Fatal: nothing is recorded at all. |
38
+ | `token storage` | A plaintext token is in a file Git tracks. Fatal. Move it to credentials or a secret manager. |
18
39
  | `ingest url` | `ingest_url` isn't a parseable HTTP(S) URL. |
40
+ | `ingest transport security` | The ingest URL is plain HTTP on a non-loopback host without `RAILWATCH_ALLOW_HTTP=true`. |
19
41
  | `ingest reachable` | `GET {ingest_url}/ingest/ping` didn't return success. Fatal. The ping carries the token, so a missing or wrong token fails this line too; fix `token` first. |
42
+
43
+ Both:
44
+
45
+ | Doctor line | What a `✗` means |
46
+ |---|---|
20
47
  | `request middleware` | `Railwatch::Middleware::Request` isn't in the stack, so requests aren't executions. |
21
48
  | `engine mounted` | `mount Railwatch::Engine, at: "/railwatch"` is missing from `config/routes.rb`; the browser beacon has nowhere to post. |
22
49
  | `deploy` | `config.deploy` is unset — records ship, charts get no deploy markers. |
23
50
  | `sample rates` | Never fails; it prints the effective rate per execution kind. |
24
51
  | `ignored record types` | Never fails; it prints what `c.ignore` is dropping. |
52
+ | `interactive sessions` | Never fails; it prints whether consoles are captured and the runner scratch paths. |
25
53
  | `kamal post-deploy hook` | `.kamal/hooks/post-deploy` is missing or doesn't mention Railwatch. Only matters if you deploy with Kamal. |
26
54
  | `browser client` | `app/frontend/lib/railwatch.ts` isn't there. Only matters for Inertia visit timing. |
27
55
  | `browser client imported` | The client exists but nothing calls `startRailwatch()` — no `startRailwatch` found in `app/frontend/entrypoints`. Visits won't report. |
@@ -33,30 +61,35 @@ isn't wired, not that the install is broken.
33
61
  **Symptom.** The environment's pages stay empty however much traffic the
34
62
  app takes.
35
63
 
36
- Work down this list. The first five are the same root cause seen from
64
+ Work down this list. Most of it is the same root cause seen from
37
65
  different angles: Railwatch decided not to record.
38
66
 
39
- **The token is missing or blank.** `Railwatch.enabled?` is
40
- `config.enabled && token.present?`. With no token the engine's
41
- `railwatch.subscribe` initializer returns early, so no subscribers and no
42
- patches are installed at all. This is by design, so the gem is inert in
43
- development. Fix: set `RAILWATCH_TOKEN`, restart, and re-run
44
- `railwatch:doctor`. The `token` line prints the first 6 characters and
45
- the length, which is enough to spot a truncated or quoted value.
67
+ **Embedded: the writer is not writing.** Run `railwatch:doctor` with the
68
+ app serving traffic. `writer process` and `last write` say whether the
69
+ writer is up and when it last wrote; the migrations lines catch a
70
+ database that was never prepared.
71
+
72
+ **Cloud: the token is missing or blank.** With `transport = :http`,
73
+ `Railwatch.enabled?` is `config.enabled && token.present?`. With no token
74
+ the engine's `railwatch.subscribe` initializer returns early, so no
75
+ subscribers and no patches are installed at all. Fix: set
76
+ `RAILWATCH_TOKEN`, restart, and re-run `railwatch:doctor`. The `token`
77
+ line prints the first 6 characters and the length, which is enough to
78
+ spot a truncated or quoted value.
46
79
 
47
- **The token is wrong.** A 401 from the ingest marks the transport
80
+ **Cloud: the token is wrong.** A 401 from the ingest marks the transport
48
81
  permanently unauthorized: no further flush is attempted for the lifetime
49
82
  of that process. Fixing the env var isn't enough. Restart the process.
50
83
  `railwatch:doctor`'s `ingest reachable` line catches this before you
51
84
  deploy.
52
85
 
53
- **`RAILWATCH_INGEST_URL` points somewhere else.** Records go where you sent
54
- them. `railwatch:status` prints the URL it is actually using. Compare it
55
- against the platform you're looking at. Self-hosting: see
86
+ **Cloud: `RAILWATCH_INGEST_URL` points somewhere else.** Records go where
87
+ you sent them. `railwatch:status` prints the URL it is actually using.
88
+ Compare it against the platform you're looking at. Self-hosting: see
56
89
  [`self-hosting.md`](self-hosting.md).
57
90
 
58
91
  **`config.enabled` is false.** `RAILWATCH_ENABLED=0` (or `false`/`no`/`off`)
59
- turns everything off with a valid token present.
92
+ turns everything off, embedded or not.
60
93
 
61
94
  **Sample rates are at zero.** `c.sample = { requests: 0.0 }` means no
62
95
  request records. So does the per-route `railwatch_never_sample` macro on
@@ -70,11 +103,11 @@ rather than a broken install.
70
103
  built. The `ignored record types` doctor line prints the list. Ignoring
71
104
  `:queries` also drops `n_plus_one`, since both key off `:queries`.
72
105
 
73
- **You're looking at the test environment.** Requiring `railwatch/rspec`
74
- (or `railwatch/minitest`) swaps the reporter's transport for an in-memory
75
- one. A suite records normally but never sends anything over the
76
- network. Independently: the health sampler, the session flusher, and the
77
- profiler all refuse to start when `Rails.env.test?`.
106
+ **You're looking at the test environment.** The spec helpers swap the
107
+ reporter's transport for an in-memory one, so a suite records normally
108
+ but never sends or stores anything. Independently: the health sampler,
109
+ the session flusher, and the profiler all refuse to start when
110
+ `Rails.env.test?`.
78
111
 
79
112
  Still nothing? Set `RAILWATCH_DEBUG=1` and restart. Internal diagnostics go
80
113
  to stderr prefixed `[railwatch]`. They never go to `Rails.logger`, so they
@@ -215,14 +248,18 @@ can't be made until it ends.
215
248
  (10,000). Past that, records are dropped and counted. The count is
216
249
  added to the reporter's drop counter so the loss is visible on the
217
250
  platform rather than silent.
218
- - `c.buffer_size` (default 10,000, the same as `MAX_RECORDS`) caps the
219
- process-wide queue between the app and the reporter thread.
220
- Oldest-dropped-first, also counted. Do not set it below `MAX_RECORDS`.
221
- An execution's tree is written to the queue in one go when it ends, so
222
- a tree larger than the queue loses its own first records. Typically
251
+ - The process-wide queue between the app and the reporter thread is
252
+ bounded by `c.buffer_bytes` (default 16 MiB) and by `c.buffer_size`
253
+ (default 10,000, the same as `MAX_RECORDS`). Oldest-dropped-first, also
254
+ counted. The byte ceiling is the one that fills: on a realistic mix of
255
+ records, 16 MiB holds about 5,000 of them, so raising `buffer_size`
256
+ changes nothing. Do not set it below `MAX_RECORDS`, though: an
257
+ execution's tree is written to the queue in one go when it ends, so a
258
+ tree larger than the queue loses its own first records. Typically
223
259
  those are the outgoing requests a long job made before it started
224
260
  writing. Keeping far more executions than before means far more
225
- records arriving at this queue. Raise it, or lower what you keep.
261
+ records arriving at this queue. Raise `buffer_bytes`, or lower what
262
+ you keep.
226
263
  - `c.profile_slow_ms` compounds it. It profiles every tail-buffering
227
264
  execution from its first line and throws away the fast ones, so the
228
265
  profiler's stack table is held alongside the record buffer.
@@ -292,7 +329,13 @@ hook below.
292
329
 
293
330
  **Cause and fix**, in the order the hook itself checks:
294
331
 
295
- - **`RAILWATCH_TOKEN` isn't exported to the hook.** The first thing
332
+ - **Embedded: `RAILWATCH_TRANSPORT=local` isn't exported to the hook.**
333
+ The hook reads the deployer's environment, not your initializer. With
334
+ that variable set it runs `bin/rails railwatch:deploy[$KAMAL_VERSION]`
335
+ in the primary container, which writes the marker to the embedded
336
+ database. Without it, the hook treats the install as a cloud one and
337
+ exits at the next check, since an embedded install has no token.
338
+ - **Cloud: `RAILWATCH_TOKEN` isn't exported to the hook.** The next thing
296
339
  `.kamal/hooks/post-deploy` does is `[ -z "$RAILWATCH_TOKEN" ] && exit 0`.
297
340
  The hook runs on the deployer machine, in your shell, not in a
298
341
  container. So a token that only exists in `.kamal/secrets` for the
@@ -313,8 +356,8 @@ it exits 0 regardless.
313
356
 
314
357
  ## Log search finds less than it should
315
358
 
316
- **Symptom.** On a Postgres-backed platform install, log search matches
317
- fewer lines and highlights nothing.
359
+ **Symptom.** On a self-hosted platform backed by Postgres, log search
360
+ matches fewer lines and highlights nothing.
318
361
 
319
362
  **Cause.** Full-text search uses SQLite's FTS5 (`logs_fts`). The
320
363
  platform checks for both a SQLite adapter *and* the `logs_fts` table.
@@ -9,10 +9,11 @@ module Railwatch
9
9
  source_root File.expand_path("templates", __dir__)
10
10
 
11
11
  desc "Creates config/initializers/railwatch.rb, a Kamal post-deploy hook, the browser client, and wires the test helpers. " \
12
- "With --local, also the two SQLite databases the in-app dashboard needs."
12
+ "By default telemetry stays in this app, in two SQLite databases, with the dashboard at /railwatch. " \
13
+ "--cloud (or any token or URL option) sends it to Railwatch Cloud instead."
13
14
 
14
- class_option :local, type: :boolean, default: false,
15
- desc: "Keep telemetry in this app and serve the dashboard at /railwatch: no token, no cloud."
15
+ class_option :cloud, type: :boolean, default: false,
16
+ desc: "Send telemetry to Railwatch Cloud instead of keeping it in this app. Implied by --prompt-token, --token-stdin, --url and --kamal-secrets."
16
17
  class_option :prompt_token, type: :boolean, default: false,
17
18
  desc: "Prompt for the ingest token without echoing it."
18
19
  class_option :token_stdin, type: :boolean, default: false,
@@ -67,9 +68,9 @@ module Railwatch
67
68
  # database is, so an app on PostgreSQL or MySQL needs the adapter gem
68
69
  # added before those files can be created.
69
70
  def ensure_sqlite3_gem
70
- return unless options[:local]
71
+ return unless local?
71
72
  return if Gem.loaded_specs.key?("sqlite3")
72
- return say("--local needs the sqlite3 gem for its two databases; add `gem \"sqlite3\"` and re-run.", :yellow) unless File.exist?("Gemfile")
73
+ return say("Embedded mode needs the sqlite3 gem for its two databases; add `gem \"sqlite3\"` and re-run.", :yellow) unless File.exist?("Gemfile")
73
74
 
74
75
  contents = File.read("Gemfile")
75
76
  unless contents.match?(/^\s*gem ["']sqlite3["']/)
@@ -89,9 +90,9 @@ module Railwatch
89
90
  # creates the tables now and migrates them after every gem update.
90
91
  # Nothing is copied into the app.
91
92
  def configure_local_databases
92
- return unless options[:local]
93
+ return unless local?
93
94
 
94
- return say("--local: no config/database.yml found; add railwatch and railwatch_telemetry databases yourself (docs/embedded.md).", :yellow) unless File.exist?("config/database.yml")
95
+ return say("No config/database.yml found; add railwatch and railwatch_telemetry databases yourself (docs/embedded.md).", :yellow) unless File.exist?("config/database.yml")
95
96
 
96
97
  contents = File.read("config/database.yml")
97
98
  updated = self.class.database_yml_with_railwatch(contents)
@@ -103,8 +104,8 @@ module Railwatch
103
104
  # The writer process: one per Puma master, forked by the gem's Puma
104
105
  # plugin, so batches are mapped and written outside the web workers.
105
106
  def configure_local_writer
106
- return unless options[:local]
107
- return say("--local: no config/puma.rb found; add `plugin :railwatch` to your Puma config yourself (docs/embedded.md).", :yellow) unless File.exist?("config/puma.rb")
107
+ return unless local?
108
+ return say("No config/puma.rb found; add `plugin :railwatch` to your Puma config yourself (docs/embedded.md).", :yellow) unless File.exist?("config/puma.rb")
108
109
 
109
110
  contents = File.read("config/puma.rb")
110
111
  updated = self.class.puma_rb_with_railwatch(contents)
@@ -166,7 +167,7 @@ module Railwatch
166
167
  # A token lands in .env only when Git confirms the file is ignored.
167
168
  # URLs are not secret and can still be written to a tracked dotenv file.
168
169
  def write_env
169
- return if options[:local]
170
+ return if local?
170
171
 
171
172
  token = resolved_token
172
173
  vars = { TOKEN_VAR => token, URL_VAR => options[:url] }.compact
@@ -219,7 +220,7 @@ module Railwatch
219
220
  # the same prepare a deploy runs, for both databases only. The host's
220
221
  # own databases are not touched, and a schema file is never written.
221
222
  def prepare_local_databases
222
- return unless options[:local]
223
+ return unless local?
223
224
  return unless File.exist?("config/database.yml")
224
225
  return if @needs_bundle
225
226
  return unless defined?(Rails) && Rails.respond_to?(:application) && Rails.application
@@ -242,21 +243,26 @@ module Railwatch
242
243
  end
243
244
 
244
245
  def show_next_steps
245
- if options[:local]
246
+ if local?
246
247
  say <<~STEPS, :green
247
248
 
248
249
  Next steps
249
- 1. Set the dashboard's HTTP Basic credentials (it is closed until
250
- you do): bin/rails railwatch:authentication:configure
250
+ 1. Restart the app and open /railwatch. In development it is open.
251
+ 2. Before production, give it a password (it is closed there until
252
+ you do): RAILS_ENV=production bin/rails railwatch:authentication:configure
251
253
  Using your own admin auth instead? See docs/embedded.md,
252
254
  Authentication (base_controller_class or a routes constraint).
253
- 2. Restart the app and open /railwatch.
254
255
  3. #{@prepared ? "Nothing else to run. Both databases were created just now and" : "Create the two databases: bin/rails db:prepare\n Then"}
255
256
  `bin/rails db:prepare` (which a deploy already runs) migrates
256
257
  them after every gem update. With `plugin :railwatch` in
257
258
  config/puma.rb Puma forks one Railwatch writer process that
258
259
  writes every batch and runs the maintenance clock, so no web
259
260
  process ever holds the telemetry database. No job worker.
261
+ 4. Optional: mirror to Railwatch Cloud for alerts delivered even
262
+ when this app is down, MCP for your AI assistant, and every
263
+ app in one place. Get a token (bin/rails railwatch:token), set
264
+ RAILWATCH_TOKEN, then c.export_enabled = true in the
265
+ initializer (docs/embedded.md).
260
266
  STEPS
261
267
  return
262
268
  end
@@ -287,7 +293,7 @@ module Railwatch
287
293
  # This process read its configuration before the initializer was
288
294
  # written, so in local mode the doctor would report an http transport
289
295
  # with no token. The databases it would check were prepared above.
290
- return if options[:local]
296
+ return if local?
291
297
  return unless defined?(Rails) && Rails.respond_to?(:application) && Rails.application
292
298
 
293
299
  say "\nbin/rails railwatch:doctor", :green
@@ -455,6 +461,16 @@ module Railwatch
455
461
 
456
462
  private
457
463
 
464
+ # Embedded unless the invocation asks for the cloud: --cloud itself, or
465
+ # an option that only means something there. A RAILWATCH_TOKEN already
466
+ # in the environment is not asking -- it is picked up when one of these
467
+ # is given, never used to choose the mode.
468
+ CLOUD_OPTIONS = %i[cloud prompt_token token_stdin url kamal_secrets].freeze
469
+
470
+ def local?
471
+ CLOUD_OPTIONS.none? { |name| options[name] }
472
+ end
473
+
458
474
  def resolved_token
459
475
  @resolved_token ||= begin
460
476
  value = if options[:prompt_token]
@@ -3,19 +3,19 @@
3
3
  # Railwatch: first-class monitoring for Rails. Every option here can also be
4
4
  # set by the RAILWATCH_* env var named in the comment.
5
5
  Railwatch.configure do |c|
6
- <% if options[:local] -%>
6
+ <% if local? -%>
7
7
  # Telemetry stays in this app's own railwatch_telemetry database and the
8
- # dashboard is served at /railwatch. No token, no cloud. Put the mount
9
- # behind your own authentication; this only names who is looking.
8
+ # dashboard is served at /railwatch. No token, no cloud. Who can open the
9
+ # dashboard is the Access block below.
10
10
  c.transport = :local # RAILWATCH_TRANSPORT
11
11
  c.ignored_request_paths += ["/railwatch", %r{\A/railwatch/}]
12
12
  # c.issue_prefix = "APP" # RAILWATCH_ISSUE_PREFIX; issue keys like APP-12
13
13
  # c.repository_url = "https://github.com/you/app" # RAILWATCH_REPOSITORY_URL; source links from stack traces
14
14
  # c.retention_days = 7 # RAILWATCH_RETENTION_DAYS; PruneTelemetryJob keeps this much
15
15
  # Access. The dashboard shows every query, log line and exception this app
16
- # records, so pick one of these. Out of the box it is HTTP Basic, on and
17
- # closed until credentials exist:
18
- # bin/rails railwatch:authentication:configure (writes Rails credentials)
16
+ # records, so pick one of these. Out of the box it is HTTP Basic: open in
17
+ # development, closed everywhere else until credentials exist:
18
+ # RAILS_ENV=production bin/rails railwatch:authentication:configure
19
19
  # or RAILWATCH_HTTP_BASIC_AUTH_USER / _PASSWORD.
20
20
  #
21
21
  # Using your own admin auth instead? Turn Basic off and say which, so the
@@ -21,6 +21,11 @@ Puma::Plugin.create do
21
21
  attr_reader :log_writer, :writer_pid
22
22
 
23
23
  POLL = 2
24
+ # How often a stopping writer is checked for, and how long a KILL is given
25
+ # to take before the pid is abandoned (a process stuck in disk I/O cannot
26
+ # die until the I/O returns, and Puma's exit should not wait for that).
27
+ REAP_POLL = 0.05
28
+ KILL_REAP = 1
24
29
 
25
30
  def start(launcher)
26
31
  @log_writer = launcher.log_writer
@@ -151,19 +156,59 @@ Puma::Plugin.create do
151
156
  log "Railwatch writer shutdown failed (#{e.class}: #{e.message})"
152
157
  end
153
158
 
154
- # TERM closes the writer's listener and it drains what it is holding; the
155
- # wait also reaps it, so a cluster master never leaves a zombie behind.
159
+ # TERM closes the writer's listener; it finishes what it is holding and
160
+ # exits on its own. Waited for with a deadline, not Process.wait, which
161
+ # has none: a writer wedged in a SQLite write or on a full disk would hold
162
+ # Puma's exit open for as long as it stayed wedged. KILL past the deadline.
163
+ # Reaped either way, so a cluster master never leaves a zombie behind.
156
164
  def stop_writer
157
165
  return unless @writer_pid
158
166
 
159
167
  Process.kill(:TERM, @writer_pid)
160
- Process.wait(@writer_pid)
168
+ return if reaped_within?(stop_timeout)
169
+
170
+ log "Railwatch writer (pid #{@writer_pid}) did not exit within #{stop_timeout}s of TERM; killing it"
171
+ Process.kill(:KILL, @writer_pid)
172
+ log "Railwatch writer (pid #{@writer_pid}) did not exit on KILL; leaving it" unless reaped_within?(KILL_REAP)
161
173
  rescue Errno::ECHILD, Errno::ESRCH
162
174
  nil
163
175
  ensure
164
176
  @writer_pid = nil
165
177
  end
166
178
 
179
+ # shutdown_timeout: the same allowance this process gives its own
180
+ # reporter, and deliberately NOT the writer's full theoretical exit time
181
+ # (a sequential SHUTDOWN_DRAIN join per worker thread, its maintenance
182
+ # join, then its own reporter shutdown -- 13s at the defaults). Waiting
183
+ # that long would buy nothing: a writer killed mid-batch loses no data,
184
+ # because the transaction rolls back and the worker retries the batch by
185
+ # id against the next writer (Writer#serve says so). Exit time spent
186
+ # waiting for the drain is spent for nothing. An idle writer is gone in
187
+ # well under a second either way.
188
+ #
189
+ # This runs from at_exit, inside the container's TERM-to-KILL grace. Under
190
+ # Kamal that is Docker's 10s default for a proxied role that has not set
191
+ # `stop_timeout`, or whatever `stop_timeout` says when it has; an app that
192
+ # wants the writer given longer can raise RAILWATCH_SHUTDOWN_TIMEOUT to
193
+ # match its grace, and this bound rises with it.
194
+ def stop_timeout
195
+ [ ::Railwatch.config.shutdown_timeout.to_f, 0.0 ].max
196
+ end
197
+
198
+ # Non-blocking waits on a short poll; Process.wait has no timeout and a
199
+ # child that never exits would hold it forever. ECHILD (the cluster's
200
+ # wait2(-1) reaped it first) propagates to stop_writer, which reads it as
201
+ # gone.
202
+ def reaped_within?(seconds)
203
+ deadline = Process.clock_gettime(Process::CLOCK_MONOTONIC) + seconds
204
+ loop do
205
+ return true if Process.waitpid(@writer_pid, Process::WNOHANG)
206
+ return false if Process.clock_gettime(Process::CLOCK_MONOTONIC) >= deadline
207
+
208
+ sleep REAP_POLL
209
+ end
210
+ end
211
+
167
212
  def log(message)
168
213
  log_writer.log(message)
169
214
  end
@@ -81,7 +81,7 @@ module Railwatch
81
81
  :max_view_renders_per_execution, :ignored_cache_key_prefixes,
82
82
  :beacon_enabled, :beacon_rate_limit, :beacon_global_rate_limit, :beacon_allowed_origins,
83
83
  :debug, :capture_default_vendor_commands,
84
- :capture_default_vendor_cache_keys, :on_unrecoverable,
84
+ :capture_default_vendor_cache_keys, :on_unrecoverable, :warn_on_data_loss,
85
85
  :capture_framework_events,
86
86
  :tail_sample_slow_ms, :failure_context, :propagate_traces, :trace_propagation_hosts,
87
87
  :health_interval, :capture_query_explain, :explain_threshold_ms,
@@ -195,6 +195,12 @@ module Railwatch
195
195
  @capture_default_vendor_cache_keys = env_bool("RAILWATCH_CAPTURE_DEFAULT_VENDOR_CACHE_KEYS", false)
196
196
  @capture_framework_events = env_bool("RAILWATCH_CAPTURE_FRAMEWORK_EVENTS", false)
197
197
  @on_unrecoverable = nil
198
+ # Off, because a gem printing into an application's own output is the
199
+ # gem changing that application's behaviour, and this one stays
200
+ # additive. Turn it on and a batch lost for good says so in one stderr
201
+ # line; leave it off and the loss shows behind RAILWATCH_DEBUG, or
202
+ # wherever on_unrecoverable routes it.
203
+ @warn_on_data_loss = env_bool("RAILWATCH_WARN_ON_DATA_LOSS", true)
198
204
  @beacon_enabled = env_bool("RAILWATCH_BEACON", true)
199
205
  # The beacon is unauthenticated and forces Railwatch.keep! for browser
200
206
  # errors, so without a ceiling anyone can spend an app's event quota
@@ -372,12 +378,21 @@ module Railwatch
372
378
  http_basic_auth_user.to_s.strip != "" && http_basic_auth_password.to_s.strip != ""
373
379
  end
374
380
 
381
+ # Development with Basic on and nothing configured is open, so a first
382
+ # run is `rails g railwatch:install` and a page, not a password step
383
+ # first. Rails already shows full error pages there. Every other
384
+ # environment stays closed until credentials exist.
385
+ def http_basic_auth_waived?
386
+ http_basic_auth_enabled && !http_basic_auth_configured? && defined?(Rails) && Rails.env.development?
387
+ end
388
+
375
389
  # Whether the request carries the configured HTTP Basic credentials.
376
390
  # False when Basic is on and nothing is configured (closed), true when
377
391
  # Basic is off (the host's base controller or routes constraint is the
378
392
  # gate then). Shared by the dashboard controller and the live channel.
379
393
  def http_basic_auth_ok?(request)
380
394
  return true unless http_basic_auth_enabled
395
+ return true if http_basic_auth_waived?
381
396
  return false unless http_basic_auth_configured?
382
397
 
383
398
  ActionController::HttpAuthentication::Basic.authenticate(request) do |user, password|
@@ -91,16 +91,27 @@ module Railwatch
91
91
  config.http_basic_auth_password ||= app.credentials.dig(:railwatch, :http_basic_auth_password)
92
92
  end
93
93
 
94
- # Two things worth one line in the log at boot, because both are
94
+ # Three things worth one line in the log at boot, because all are
95
95
  # invisible until something is already wrong: a Rails/json pair that
96
- # cannot decode, and an embedded dashboard with nothing declared in
97
- # front of it.
96
+ # cannot decode, an embedded dashboard with nothing declared in front of
97
+ # it, and one that is closed to everyone because Basic has no
98
+ # credentials outside development.
98
99
  initializer "railwatch.warnings", after: :load_config_initializers do
99
100
  config.after_initialize do
100
101
  next unless Railwatch.enabled?
101
102
 
102
103
  Rails.logger.warn("[railwatch] #{Railwatch::JsonCompat.advice}") if Railwatch::JsonCompat.broken?
103
104
 
105
+ if Railwatch.config.local? && Railwatch.config.dashboard_gate == :basic &&
106
+ !Railwatch.config.http_basic_auth_configured? && !Rails.env.local?
107
+ Rails.logger.warn(
108
+ "[railwatch] the dashboard is closed: HTTP Basic is on and no credentials are configured for " \
109
+ "#{Rails.env}, so every request to it is 401. Run `RAILS_ENV=#{Rails.env} bin/rails " \
110
+ "railwatch:authentication:configure`, or gate it with your own auth and set " \
111
+ "`c.http_basic_auth_enabled = false` (docs/embedded.md)."
112
+ )
113
+ end
114
+
104
115
  if Railwatch.config.local? && Railwatch.config.dashboard_gate == :undeclared && !Rails.env.local?
105
116
  Rails.logger.warn(
106
117
  "[railwatch] the dashboard at the engine's mount has no gate this gem can see: HTTP Basic is off and " \
@@ -100,10 +100,17 @@ module Railwatch
100
100
 
101
101
  def flush
102
102
  ensure_process!
103
- @flush_mutex.synchronize do
103
+ deferred = nil
104
+ result = @flush_mutex.synchronize do
105
+ @deferred_notifications = []
104
106
  update_backpressure
105
107
  deliver_buffer
108
+ ensure
109
+ deferred = @deferred_notifications
110
+ @deferred_notifications = nil
106
111
  end
112
+ deferred&.each { |error| Railwatch.notify_unrecoverable(error) }
113
+ result
107
114
  end
108
115
 
109
116
  def ensure_thread
@@ -278,6 +285,19 @@ module Railwatch
278
285
  end
279
286
  end
280
287
 
288
+ # Everything reachable from deliver_buffer runs inside @flush_mutex, so
289
+ # the loss it reports cannot be handed to the application there. Ruby's
290
+ # Mutex is not reentrant: the documented callback is Rails.error.report,
291
+ # whose subscriber records the error as an exception and can end up asking
292
+ # this same reporter to flush -- "ThreadError: deadlock; recursive
293
+ # locking", rescued by notify_unrecoverable and so a callback cut off
294
+ # halfway, reporting nothing. Collected here and dispatched by flush once
295
+ # the lock is released. retain already does this for @mutex; @flush_mutex
296
+ # is the outer one it still sat inside.
297
+ def defer_notification(error)
298
+ @deferred_notifications ? @deferred_notifications << error : Railwatch.notify_unrecoverable(error)
299
+ end
300
+
281
301
  def deliver_buffer
282
302
  # A 401 was reported once, when the transport first saw it; after
283
303
  # that the token is wrong until the process restarts, and repeating
@@ -337,7 +357,7 @@ module Railwatch
337
357
  rescue StandardError => e
338
358
  result = Transport::Http::Result.new(ok: false, error: "#{e.class}: #{e.message}")
339
359
  batch&.records&.any? ? retain(batch, result) : delivery_succeeded
340
- Railwatch.notify_unrecoverable(e)
360
+ defer_notification(e)
341
361
  result
342
362
  ensure
343
363
  in_flight(0, 0, 0, 0)
@@ -350,7 +370,7 @@ module Railwatch
350
370
  end
351
371
 
352
372
  def retain(batch, result)
353
- @mutex.synchronize do
373
+ gave_up = @mutex.synchronize do
354
374
  @in_flight_records = 0
355
375
  @in_flight_dropped = 0
356
376
  @in_flight_bytes = 0
@@ -362,17 +382,7 @@ module Railwatch
362
382
  @retry_attempt = 0
363
383
  @retry_at = nil
364
384
  Railwatch.debug { "gave up on a batch of #{batch.records.size} records after #{MAX_RETRY_ATTEMPTS} retries (#{result.error || result.status}); dropped and counted" }
365
- # Losing a batch is not a debug-level event: with an ingest (or an
366
- # embedded writer) that never comes back this is the only place the
367
- # loss is ever reported, and the dropped counter it leaves behind
368
- # rides on the NEXT successful delivery, which may never happen.
369
- Railwatch.notify_unrecoverable(
370
- DeliveryError.new("Railwatch dropped #{batch.records.size} records after #{MAX_RETRY_ATTEMPTS} failed delivery attempts: " \
371
- "#{result.error || result.status}",
372
- status: result.status, records: batch.records.size, bytes: batch.bytes,
373
- dropped: batch.dropped, dropped_bytes: batch.dropped_bytes)
374
- )
375
- next
385
+ next true
376
386
  end
377
387
  @retry_batch = batch
378
388
  delay = retry_delay(@retry_attempt)
@@ -382,7 +392,32 @@ module Railwatch
382
392
  "retained #{batch.records.size} records after retryable delivery failure " \
383
393
  "(#{result.error || result.status}); retry #{@retry_attempt} in #{delay.round(3)}s"
384
394
  end
395
+ false
385
396
  end
397
+ return unless gave_up
398
+
399
+ # Losing a batch is not a debug-level event: with an ingest (or an
400
+ # embedded writer) that never comes back this is the only place the
401
+ # loss is ever reported, and the dropped counter it leaves behind
402
+ # rides on the NEXT successful delivery, which may never happen.
403
+ #
404
+ # Reported here, after the lock is released, never inside it. The
405
+ # callback is the app's: the documented one is Rails.error.report,
406
+ # whose subscriber records the error as an exception and so writes
407
+ # straight back into this reporter -- which needs @mutex to arm the
408
+ # thread or ask for a flush, and under the lock that was
409
+ # "ThreadError: deadlock; recursive locking" and a callback cut off
410
+ # halfway. A slow callback under the lock was worse: shutdown's own
411
+ # @mutex.synchronize and every request thread's write_now sat behind
412
+ # it for as long as it took. notify_unsent and delivery_rejected
413
+ # already call out unlocked; this was the one that did not.
414
+ defer_notification(
415
+ DeliveryError.new("Railwatch dropped #{batch.records.size} #{batch.records.size == 1 ? "record" : "records"} " \
416
+ "after #{MAX_RETRY_ATTEMPTS + 1} failed delivery attempts: " \
417
+ "#{result.error || result.status}",
418
+ status: result.status, records: batch.records.size, bytes: batch.bytes,
419
+ dropped: batch.dropped, dropped_bytes: batch.dropped_bytes)
420
+ )
386
421
  end
387
422
 
388
423
  def discard_unauthorized
@@ -406,7 +441,7 @@ module Railwatch
406
441
  def delivery_rejected(batch, result)
407
442
  delivery_succeeded
408
443
  detail = result.error.to_s.empty? ? "HTTP #{result.status}" : result.error
409
- Railwatch.notify_unrecoverable(
444
+ defer_notification(
410
445
  DeliveryError.new("Railwatch ingest permanently rejected #{batch.records.size} records: #{detail}",
411
446
  status: result.status, records: batch.records.size, bytes: batch.bytes,
412
447
  dropped: batch.dropped, dropped_bytes: batch.dropped_bytes)
@@ -550,7 +585,8 @@ module Railwatch
550
585
  return unless should_notify
551
586
 
552
587
  Railwatch.notify_unrecoverable(
553
- DeliveryError.new("Railwatch #{reason} with #{records} unsent records retained in memory (#{bytes} bytes)",
588
+ DeliveryError.new("Railwatch #{reason} with #{records} unsent #{records == 1 ? "record" : "records"} " \
589
+ "retained in memory (#{bytes} bytes)",
554
590
  records: records, bytes: bytes, dropped: dropped, dropped_bytes: dropped_bytes)
555
591
  )
556
592
  end
@@ -8,9 +8,10 @@ require "json"
8
8
 
9
9
  module Railwatch
10
10
  module Transport
11
- # POSTs gzip NDJSON batches to the platform. Each call retries one raised
12
- # error or 5xx response, then returns a classified, non-raising result;
13
- # Reporter owns retention and backoff between calls. A 401 marks the
11
+ # POSTs gzip NDJSON batches to the platform. Each call makes exactly one
12
+ # attempt and returns a classified, non-raising result; Reporter owns
13
+ # retention, the retry ladder, and backoff between calls, so a retry here
14
+ # would multiply into its schedule rather than add to it. A 401 marks the
14
15
  # transport unauthorized so no further requests are made.
15
16
  class Http
16
17
  # The request headers a delivery carries besides the body. The receiver
@@ -87,17 +88,11 @@ module Railwatch
87
88
  dropped += over_cap
88
89
  dropped_bytes += over_cap_bytes
89
90
  end
90
- attempt = 0
91
91
  begin
92
- attempt += 1
93
92
  result = parse(post(body, dropped, dropped_bytes, backpressure_factor, batch_id), expected_count: sent)
94
- if attempt < 2 && (500..599).cover?(result.status)
95
- result = parse(post(body, dropped, dropped_bytes, backpressure_factor, batch_id), expected_count: sent)
96
- end
97
93
  apply_status_policy(result)
98
94
  result
99
95
  rescue StandardError => e
100
- retry if attempt < 2
101
96
  Result.new(ok: false, error: "#{e.class}: #{e.message}")
102
97
  end
103
98
  end
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Railwatch
4
- VERSION = "0.5.1"
4
+ VERSION = "0.6.1"
5
5
  end