railwatch 0.5.1 → 0.6.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/AGENTS.md +45 -25
- data/CHANGELOG.md +147 -0
- data/README.md +50 -31
- data/app/controllers/railwatch/dashboard_controller.rb +1 -0
- data/app/jobs/railwatch/rollup_job.rb +3 -1
- data/app/models/railwatch/application_record.rb +2 -2
- data/app/models/railwatch/ingest/rollup_absorber.rb +1 -1
- data/app/models/railwatch/telemetry_record.rb +2 -2
- data/docs/ai-and-mcp.md +9 -3
- data/docs/configuration.md +97 -40
- data/docs/embedded.md +96 -43
- data/docs/faq.md +28 -15
- data/docs/getting-started.md +121 -42
- data/docs/records.md +13 -10
- data/docs/replacing-nightwatch.md +16 -14
- data/docs/replacing-sentry.md +19 -10
- data/docs/security.md +24 -3
- data/docs/self-hosting.md +9 -1
- data/docs/testing.md +14 -4
- data/docs/troubleshooting.md +72 -29
- data/lib/generators/railwatch/install/install_generator.rb +32 -16
- data/lib/generators/railwatch/install/templates/initializer.rb.tt +6 -6
- data/lib/puma/plugin/railwatch.rb +48 -3
- data/lib/railwatch/configuration.rb +16 -1
- data/lib/railwatch/engine.rb +14 -3
- data/lib/railwatch/reporter.rb +52 -16
- data/lib/railwatch/transport/http.rb +4 -9
- data/lib/railwatch/version.rb +1 -1
- data/lib/railwatch.rb +23 -1
- data/lib/tasks/railwatch_tasks.rake +11 -4
- data/llms.txt +21 -14
- data/public/railwatch/assets/{app-layout-DDyQa72H.js → app-layout-Zc0v-hYh.js} +1 -1
- data/public/railwatch/assets/{app-wordmark-o9CODKP0.js → app-wordmark-BA_60AVb.js} +1 -1
- data/public/railwatch/assets/{appearance-BwuCXabr.js → appearance-CcfP9tZ7.js} +1 -1
- data/public/railwatch/assets/application-BrN3Sz94.css +1 -0
- data/public/railwatch/assets/{arrow-up-C6PxDiY3.js → arrow-up-CLQ-7heQ.js} +1 -1
- data/public/railwatch/assets/{auth-layout-BRt8MGFD.js → auth-layout-C05gEBIQ.js} +1 -1
- data/public/railwatch/assets/{badge-CAxXV8za.js → badge-DRae8XwK.js} +1 -1
- data/public/railwatch/assets/{braces-DgomTCNf.js → braces-rSYydpLY.js} +1 -1
- data/public/railwatch/assets/{card-cAtqCxWl.js → card-DPjFKfen.js} +1 -1
- data/public/railwatch/assets/{chart-BBeBkkNa.js → chart-DWh7l8yM.js} +1 -1
- data/public/railwatch/assets/{chart-hover-B1M9jc0y.js → chart-hover-CfoZUY4J.js} +1 -1
- data/public/railwatch/assets/{chart-panel-DUQTz_C8.js → chart-panel-CC49WTQL.js} +1 -1
- data/public/railwatch/assets/{checkbox-CmhMHWZO.js → checkbox-DPkLUiwM.js} +1 -1
- data/public/railwatch/assets/{code-DESvxyTj.js → code-CLmYS6FU.js} +1 -1
- data/public/railwatch/assets/{copy-block-BkSU5832.js → copy-block-CGcXxp8J.js} +1 -1
- data/public/railwatch/assets/{copy-id-D03GhN9F.js → copy-id-vBYHQwxg.js} +1 -1
- data/public/railwatch/assets/{cursor-load-more-CRyuMeQb.js → cursor-load-more-Ddwxe5dZ.js} +1 -1
- data/public/railwatch/assets/{data-table-BIlt7Rtm.js → data-table-CzKTEE-O.js} +1 -1
- data/public/railwatch/assets/{edit-O0NSBWxo.js → edit-B1kmWkzd.js} +1 -1
- data/public/railwatch/assets/{edit-DJ0D0wHN.js → edit-D9cx4pbG.js} +1 -1
- data/public/railwatch/assets/{edit-Bb6MKoe4.js → edit-yv6j9p-V.js} +1 -1
- data/public/railwatch/assets/{empty-state-C38il627.js → empty-state-CW4wclK_.js} +1 -1
- data/public/railwatch/assets/{env-layout-REF7OM4q.js → env-layout-Kz7wks1x.js} +1 -1
- data/public/railwatch/assets/{execution-path-FYLq1TwC.js → execution-path-FAuIzOXB.js} +1 -1
- data/public/railwatch/assets/{filter-bar-CYog9Alp.js → filter-bar-CJDjWFib.js} +1 -1
- data/public/railwatch/assets/{flamegraph-DSs69foN.js → flamegraph-EaGkP2NT.js} +1 -1
- data/public/railwatch/assets/{frames-BUi2J5Mk.js → frames-zZuIaNH9.js} +1 -1
- data/public/railwatch/assets/{google-sign-in-button-BQiIKFdd.js → google-sign-in-button-2_zgbgVy.js} +1 -1
- data/public/railwatch/assets/{index-ZOGOB8SA.js → index-B49SWz7K.js} +1 -1
- data/public/railwatch/assets/{index-DqTFTP8p.js → index-BHlY4wKe.js} +1 -1
- data/public/railwatch/assets/{index-DrcKVG2f.js → index-BZpPtyFY.js} +1 -1
- data/public/railwatch/assets/{index-Dh4IRLFI.js → index-BeVK6jCL.js} +1 -1
- data/public/railwatch/assets/{index-umIAl-pL.js → index-BfbSo01U.js} +1 -1
- data/public/railwatch/assets/{index-BoUBioBP.js → index-BfgncAv6.js} +1 -1
- data/public/railwatch/assets/{index-C3A_9imx.js → index-BhNszK1k.js} +1 -1
- data/public/railwatch/assets/{index-tpz-OGUP.js → index-Bu01uWvw.js} +1 -1
- data/public/railwatch/assets/{index-CGs4m_fa.js → index-C1s_hK3p.js} +1 -1
- data/public/railwatch/assets/{index-ZSZg9rtq.js → index-C8Cggnbw.js} +1 -1
- data/public/railwatch/assets/{index-so4lRrRq.js → index-CBip6V4z.js} +1 -1
- data/public/railwatch/assets/{index-DvjY3dPD.js → index-CMJGss5R.js} +1 -1
- data/public/railwatch/assets/{index-DtHmuB9Q.js → index-CRo3yK20.js} +1 -1
- data/public/railwatch/assets/{index-r0tSIplE.js → index-CWB_p2J8.js} +1 -1
- data/public/railwatch/assets/{index-CICUIFHL.js → index-CWHpndGc.js} +1 -1
- data/public/railwatch/assets/{index-CFFpnzIS.js → index-Ca_S4Sc3.js} +1 -1
- data/public/railwatch/assets/{index-DSvlZVWG.js → index-CanPDDOa.js} +1 -1
- data/public/railwatch/assets/{index-CrZ3vHDL.js → index-Cie90Yat.js} +1 -1
- data/public/railwatch/assets/{index-FhUaPPab.js → index-CiepQ_pR.js} +1 -1
- data/public/railwatch/assets/{index-C_upSl_k.js → index-Cm1uCIGN.js} +1 -1
- data/public/railwatch/assets/{index-sTYvcbkh.js → index-CzitnnSC.js} +1 -1
- data/public/railwatch/assets/{index-C7OtLq_3.js → index-DEUfClv3.js} +1 -1
- data/public/railwatch/assets/index-DJKwo-mI.js +1 -0
- data/public/railwatch/assets/{index-DDI_Zx5V.js → index-DK6y0YHp.js} +1 -1
- data/public/railwatch/assets/{index-C-PmdhXA.js → index-DMNPpLH9.js} +1 -1
- data/public/railwatch/assets/{index-BaR1U9An.js → index-DZO1mSfX.js} +1 -1
- data/public/railwatch/assets/{index-CsoN51vW.js → index-Db5wj3M5.js} +1 -1
- data/public/railwatch/assets/{index-BiiyMcA0.js → index-Dcy5WktB.js} +1 -1
- data/public/railwatch/assets/{index-8-hnAhOD.js → index-DvG0-7Lx.js} +1 -1
- data/public/railwatch/assets/{index-QpTtwFwu.js → index-Dvj1wuka.js} +1 -1
- data/public/railwatch/assets/{index-DW2CBbxU.js → index-KyZX46qX.js} +1 -1
- data/public/railwatch/assets/{index-BeOh2t_S.js → index-Ze-KP-sl.js} +1 -1
- data/public/railwatch/assets/{index-CiPo4Gob.js → index-mhRbWLBM.js} +1 -1
- data/public/railwatch/assets/{index-CpkI015n.js → index-x099JL5f.js} +1 -1
- data/public/railwatch/assets/{index-JdCVBrw8.js → index-x28zb_nf.js} +1 -1
- data/public/railwatch/assets/{inertia-DLew8ZNx.js → inertia-Cuyz2ZHO.js} +2 -2
- data/public/railwatch/assets/{input-error-cvM6_Jht.js → input-error-hog6gGxg.js} +1 -1
- data/public/railwatch/assets/{json-viewer-D922McGi.js → json-viewer-DH2W9HXf.js} +1 -1
- data/public/railwatch/assets/{klass-CrwICqN8.js → klass-DHbDelLk.js} +1 -1
- data/public/railwatch/assets/{label-GWl7I6sf.js → label-DdCBgiUn.js} +1 -1
- data/public/railwatch/assets/{layout-0ZAnD3zl.js → layout-Cueyl7c5.js} +1 -1
- data/public/railwatch/assets/{live-dot-D1n_BreY.js → live-dot-BZgYTYdt.js} +1 -1
- data/public/railwatch/assets/{nav-DPxr1NNC.js → nav-BSSGObDZ.js} +1 -1
- data/public/railwatch/assets/{new-D-ZzUK9a.js → new-B8FSb8Bl.js} +1 -1
- data/public/railwatch/assets/{new-DEVkYv-z.js → new-C39_v2Ll.js} +1 -1
- data/public/railwatch/assets/{new-Cdl6pqST.js → new-DVOPwaC5.js} +1 -1
- data/public/railwatch/assets/{new-DHAHDrN7.js → new-Dd40nvJR.js} +1 -1
- data/public/railwatch/assets/{new-Dz4lZf1L.js → new-SxnYe1SE.js} +1 -1
- data/public/railwatch/assets/{new-GMrRFurX.js → new-eP5vKD3Y.js} +1 -1
- data/public/railwatch/assets/onboarding-CUpZl5KB.js +1 -0
- data/public/railwatch/assets/{origin-identity-Bk9yHWZ1.js → origin-identity-BYt2uiuo.js} +1 -1
- data/public/railwatch/assets/{percentile-picker-DfSx9yJO.js → percentile-picker-CeQgllxD.js} +1 -1
- data/public/railwatch/assets/{relative-time-CjIjb8Lg.js → relative-time-D5UbF4oO.js} +1 -1
- data/public/railwatch/assets/{release-health-4b3tivEf.js → release-health-uP-GF3_V.js} +1 -1
- data/public/railwatch/assets/{route-C_5BUtHK.js → route-DQAY8JEr.js} +1 -1
- data/public/railwatch/assets/{segmented-BgbT3wZa.js → segmented-BeIe4uqk.js} +1 -1
- data/public/railwatch/assets/{select-DmunxCKE.js → select-2R16457Z.js} +1 -1
- data/public/railwatch/assets/{separator-BXzEdZ_8.js → separator-D5I0UCB5.js} +1 -1
- data/public/railwatch/assets/series-chart-gnrzhmM6.js +1 -0
- data/public/railwatch/assets/show-2BkeNRUC.js +2 -0
- data/public/railwatch/assets/{show-mU38uGTg.js → show-B0X1hRQH.js} +1 -1
- data/public/railwatch/assets/{show-CAl7xcex.js → show-BKUV5l5q.js} +1 -1
- data/public/railwatch/assets/{show-DnR1Dnjd.js → show-BQJD_CsF.js} +1 -1
- data/public/railwatch/assets/{show-Dn-GwFZL.js → show-BhmD6Sxp.js} +1 -1
- data/public/railwatch/assets/{show-Y74rM0VT.js → show-C3KcnVo2.js} +1 -1
- data/public/railwatch/assets/{show-Dily73Xk.js → show-C95cHd06.js} +1 -1
- data/public/railwatch/assets/{show-DXs4deaC.js → show-CdB4uVQS.js} +1 -1
- data/public/railwatch/assets/{show-DI8IhNUH.js → show-CoCqcVIp.js} +1 -1
- data/public/railwatch/assets/{show-vQ4bndYD.js → show-D2LqjEsU.js} +1 -1
- data/public/railwatch/assets/{show-DcpTFiLi.js → show-DNpnKyTh.js} +1 -1
- data/public/railwatch/assets/{show-B2zLAW83.js → show-DOlTxbig.js} +1 -1
- data/public/railwatch/assets/{show-SvLOcPrx.js → show-DQCkttL8.js} +1 -1
- data/public/railwatch/assets/{show-BLpWUHWD.js → show-DcxesipC.js} +1 -1
- data/public/railwatch/assets/{show-DYskfl3-.js → show-IGJoc_X0.js} +1 -1
- data/public/railwatch/assets/{show-DSP9Cq_C.js → show-YsNpqbcI.js} +1 -1
- data/public/railwatch/assets/{show-DlRVS18-.js → show-qNV6H8SH.js} +1 -1
- data/public/railwatch/assets/{sort-header-Dcq9bzmo.js → sort-header-mEJUM3Av.js} +1 -1
- data/public/railwatch/assets/{sparkline-cell-BON3qQUB.js → sparkline-cell-Dwg7awwh.js} +1 -1
- data/public/railwatch/assets/{stat-DFEyFxkO.js → stat-ZtxGU8lE.js} +1 -1
- data/public/railwatch/assets/{status-badge-BaUKP7Yo.js → status-badge-CH5P-Xkj.js} +1 -1
- data/public/railwatch/assets/{tenant-path-DPZPc985.js → tenant-path-CQoP-BeF.js} +1 -1
- data/public/railwatch/assets/{text-link-BO77t9Xk.js → text-link-DHJ8BXx5.js} +1 -1
- data/public/railwatch/assets/{textarea-DTqrCiV0.js → textarea-BeQtQyl5.js} +1 -1
- data/public/railwatch/assets/{timeline-D5rJ0es2.js → timeline-CaKQVu48.js} +1 -1
- data/public/railwatch/assets/{transition-DMIrZVth.js → transition-ksDpqhKJ.js} +1 -1
- data/public/railwatch/assets/{use-clipboard-ByoUGQqA.js → use-clipboard-DNefo-ky.js} +1 -1
- data/public/railwatch/assets/{use-live-D7xKz2ma.js → use-live-DMTuhKfB.js} +1 -1
- data/public/railwatch/manifest.json +1286 -1286
- metadata +116 -116
- data/public/railwatch/assets/application-B7h1MIhi.css +0 -1
- data/public/railwatch/assets/index-CFRLPs4J.js +0 -1
- data/public/railwatch/assets/onboarding-D1vwaHYT.js +0 -1
- data/public/railwatch/assets/series-chart-Xf49v9cv.js +0 -1
- data/public/railwatch/assets/show-SHwZjXb7.js +0 -2
data/docs/troubleshooting.md
CHANGED
|
@@ -8,20 +8,48 @@ below are keyed to those lines.
|
|
|
8
8
|
bin/rails railwatch:doctor
|
|
9
9
|
```
|
|
10
10
|
|
|
11
|
-
The
|
|
12
|
-
|
|
11
|
+
The first lines depend on the transport. An embedded install checks its
|
|
12
|
+
databases, writer and dashboard; a cloud install checks its token and
|
|
13
|
+
ingest host. The task exits non-zero only on the lines marked fatal
|
|
14
|
+
below. Everything else is informational: a `✗` there means a feature
|
|
13
15
|
isn't wired, not that the install is broken.
|
|
14
16
|
|
|
17
|
+
Embedded (`transport = :local`):
|
|
18
|
+
|
|
19
|
+
| Doctor line | What a `✗` means |
|
|
20
|
+
|---|---|
|
|
21
|
+
| `export` | Only shown with `export_enabled` on. Export is on and cannot work: no token, no URL, a plain-HTTP URL, or an unsupported policy. Fatal. |
|
|
22
|
+
| `export destination` | The export queue is blocked or deferred; the line gives the reason. `bin/rails railwatch:export:rebind` clears a credential block. |
|
|
23
|
+
| `railwatch database`, `railwatch_telemetry database` | The database is missing from `config/database.yml` for this environment. Fatal. Re-run the install generator. |
|
|
24
|
+
| `railwatch migrations`, `railwatch_telemetry migrations` | Migrations are pending. Fatal. Run `bin/rails db:prepare`. |
|
|
25
|
+
| `telemetry disk` | The telemetry database is not in incremental auto-vacuum, so pruning never shrinks the file. See [Embedded mode](embedded.md#giving-the-disk-back). |
|
|
26
|
+
| `maintenance` | No maintenance tick in the last ten minutes. Expected when the app is stopped: the clock runs in the app's processes, not in rake. |
|
|
27
|
+
| `writer process` | The writer socket is not answering, `plugin :railwatch` is missing from `config/puma.rb`, or the socket path is over Linux's 108-byte limit. Expected when the app is stopped. |
|
|
28
|
+
| `last write` | Shown when the writer answers: nothing written in five minutes. With the app serving traffic, the writer is stuck or workers are not reaching it. |
|
|
29
|
+
| `dashboard access` | HTTP Basic is on with no credentials outside development, so every page is 401; or Basic is off and nothing else is declared. Run `bin/rails railwatch:authentication:configure`, or see [Embedded mode](embedded.md#authentication). |
|
|
30
|
+
| `json compatibility` | The installed `json` gem cannot decode on this Rails; see [below](#binjobs-dies-in-a-loop-with-wrong-number-of-arguments-given-2-expected-1). |
|
|
31
|
+
| `recurring.yml` | `config/recurring.yml` still lists `Railwatch::*` jobs from a pre-release. Remove them. |
|
|
32
|
+
|
|
33
|
+
Cloud (`transport = :http`):
|
|
34
|
+
|
|
15
35
|
| Doctor line | What a `✗` means |
|
|
16
36
|
|---|---|
|
|
17
37
|
| `token` | `RAILWATCH_TOKEN` is unset or empty. Fatal: nothing is recorded at all. |
|
|
38
|
+
| `token storage` | A plaintext token is in a file Git tracks. Fatal. Move it to credentials or a secret manager. |
|
|
18
39
|
| `ingest url` | `ingest_url` isn't a parseable HTTP(S) URL. |
|
|
40
|
+
| `ingest transport security` | The ingest URL is plain HTTP on a non-loopback host without `RAILWATCH_ALLOW_HTTP=true`. |
|
|
19
41
|
| `ingest reachable` | `GET {ingest_url}/ingest/ping` didn't return success. Fatal. The ping carries the token, so a missing or wrong token fails this line too; fix `token` first. |
|
|
42
|
+
|
|
43
|
+
Both:
|
|
44
|
+
|
|
45
|
+
| Doctor line | What a `✗` means |
|
|
46
|
+
|---|---|
|
|
20
47
|
| `request middleware` | `Railwatch::Middleware::Request` isn't in the stack, so requests aren't executions. |
|
|
21
48
|
| `engine mounted` | `mount Railwatch::Engine, at: "/railwatch"` is missing from `config/routes.rb`; the browser beacon has nowhere to post. |
|
|
22
49
|
| `deploy` | `config.deploy` is unset — records ship, charts get no deploy markers. |
|
|
23
50
|
| `sample rates` | Never fails; it prints the effective rate per execution kind. |
|
|
24
51
|
| `ignored record types` | Never fails; it prints what `c.ignore` is dropping. |
|
|
52
|
+
| `interactive sessions` | Never fails; it prints whether consoles are captured and the runner scratch paths. |
|
|
25
53
|
| `kamal post-deploy hook` | `.kamal/hooks/post-deploy` is missing or doesn't mention Railwatch. Only matters if you deploy with Kamal. |
|
|
26
54
|
| `browser client` | `app/frontend/lib/railwatch.ts` isn't there. Only matters for Inertia visit timing. |
|
|
27
55
|
| `browser client imported` | The client exists but nothing calls `startRailwatch()` — no `startRailwatch` found in `app/frontend/entrypoints`. Visits won't report. |
|
|
@@ -33,30 +61,35 @@ isn't wired, not that the install is broken.
|
|
|
33
61
|
**Symptom.** The environment's pages stay empty however much traffic the
|
|
34
62
|
app takes.
|
|
35
63
|
|
|
36
|
-
Work down this list.
|
|
64
|
+
Work down this list. Most of it is the same root cause seen from
|
|
37
65
|
different angles: Railwatch decided not to record.
|
|
38
66
|
|
|
39
|
-
**
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
67
|
+
**Embedded: the writer is not writing.** Run `railwatch:doctor` with the
|
|
68
|
+
app serving traffic. `writer process` and `last write` say whether the
|
|
69
|
+
writer is up and when it last wrote; the migrations lines catch a
|
|
70
|
+
database that was never prepared.
|
|
71
|
+
|
|
72
|
+
**Cloud: the token is missing or blank.** With `transport = :http`,
|
|
73
|
+
`Railwatch.enabled?` is `config.enabled && token.present?`. With no token
|
|
74
|
+
the engine's `railwatch.subscribe` initializer returns early, so no
|
|
75
|
+
subscribers and no patches are installed at all. Fix: set
|
|
76
|
+
`RAILWATCH_TOKEN`, restart, and re-run `railwatch:doctor`. The `token`
|
|
77
|
+
line prints the first 6 characters and the length, which is enough to
|
|
78
|
+
spot a truncated or quoted value.
|
|
46
79
|
|
|
47
|
-
**
|
|
80
|
+
**Cloud: the token is wrong.** A 401 from the ingest marks the transport
|
|
48
81
|
permanently unauthorized: no further flush is attempted for the lifetime
|
|
49
82
|
of that process. Fixing the env var isn't enough. Restart the process.
|
|
50
83
|
`railwatch:doctor`'s `ingest reachable` line catches this before you
|
|
51
84
|
deploy.
|
|
52
85
|
|
|
53
|
-
|
|
54
|
-
them. `railwatch:status` prints the URL it is actually using.
|
|
55
|
-
against the platform you're looking at. Self-hosting: see
|
|
86
|
+
**Cloud: `RAILWATCH_INGEST_URL` points somewhere else.** Records go where
|
|
87
|
+
you sent them. `railwatch:status` prints the URL it is actually using.
|
|
88
|
+
Compare it against the platform you're looking at. Self-hosting: see
|
|
56
89
|
[`self-hosting.md`](self-hosting.md).
|
|
57
90
|
|
|
58
91
|
**`config.enabled` is false.** `RAILWATCH_ENABLED=0` (or `false`/`no`/`off`)
|
|
59
|
-
turns everything off
|
|
92
|
+
turns everything off, embedded or not.
|
|
60
93
|
|
|
61
94
|
**Sample rates are at zero.** `c.sample = { requests: 0.0 }` means no
|
|
62
95
|
request records. So does the per-route `railwatch_never_sample` macro on
|
|
@@ -70,11 +103,11 @@ rather than a broken install.
|
|
|
70
103
|
built. The `ignored record types` doctor line prints the list. Ignoring
|
|
71
104
|
`:queries` also drops `n_plus_one`, since both key off `:queries`.
|
|
72
105
|
|
|
73
|
-
**You're looking at the test environment.**
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
106
|
+
**You're looking at the test environment.** The spec helpers swap the
|
|
107
|
+
reporter's transport for an in-memory one, so a suite records normally
|
|
108
|
+
but never sends or stores anything. Independently: the health sampler,
|
|
109
|
+
the session flusher, and the profiler all refuse to start when
|
|
110
|
+
`Rails.env.test?`.
|
|
78
111
|
|
|
79
112
|
Still nothing? Set `RAILWATCH_DEBUG=1` and restart. Internal diagnostics go
|
|
80
113
|
to stderr prefixed `[railwatch]`. They never go to `Rails.logger`, so they
|
|
@@ -215,14 +248,18 @@ can't be made until it ends.
|
|
|
215
248
|
(10,000). Past that, records are dropped and counted. The count is
|
|
216
249
|
added to the reporter's drop counter so the loss is visible on the
|
|
217
250
|
platform rather than silent.
|
|
218
|
-
-
|
|
219
|
-
|
|
220
|
-
Oldest-dropped-first, also
|
|
221
|
-
|
|
222
|
-
|
|
251
|
+
- The process-wide queue between the app and the reporter thread is
|
|
252
|
+
bounded by `c.buffer_bytes` (default 16 MiB) and by `c.buffer_size`
|
|
253
|
+
(default 10,000, the same as `MAX_RECORDS`). Oldest-dropped-first, also
|
|
254
|
+
counted. The byte ceiling is the one that fills: on a realistic mix of
|
|
255
|
+
records, 16 MiB holds about 5,000 of them, so raising `buffer_size`
|
|
256
|
+
changes nothing. Do not set it below `MAX_RECORDS`, though: an
|
|
257
|
+
execution's tree is written to the queue in one go when it ends, so a
|
|
258
|
+
tree larger than the queue loses its own first records. Typically
|
|
223
259
|
those are the outgoing requests a long job made before it started
|
|
224
260
|
writing. Keeping far more executions than before means far more
|
|
225
|
-
records arriving at this queue. Raise
|
|
261
|
+
records arriving at this queue. Raise `buffer_bytes`, or lower what
|
|
262
|
+
you keep.
|
|
226
263
|
- `c.profile_slow_ms` compounds it. It profiles every tail-buffering
|
|
227
264
|
execution from its first line and throws away the fast ones, so the
|
|
228
265
|
profiler's stack table is held alongside the record buffer.
|
|
@@ -292,7 +329,13 @@ hook below.
|
|
|
292
329
|
|
|
293
330
|
**Cause and fix**, in the order the hook itself checks:
|
|
294
331
|
|
|
295
|
-
-
|
|
332
|
+
- **Embedded: `RAILWATCH_TRANSPORT=local` isn't exported to the hook.**
|
|
333
|
+
The hook reads the deployer's environment, not your initializer. With
|
|
334
|
+
that variable set it runs `bin/rails railwatch:deploy[$KAMAL_VERSION]`
|
|
335
|
+
in the primary container, which writes the marker to the embedded
|
|
336
|
+
database. Without it, the hook treats the install as a cloud one and
|
|
337
|
+
exits at the next check, since an embedded install has no token.
|
|
338
|
+
- **Cloud: `RAILWATCH_TOKEN` isn't exported to the hook.** The next thing
|
|
296
339
|
`.kamal/hooks/post-deploy` does is `[ -z "$RAILWATCH_TOKEN" ] && exit 0`.
|
|
297
340
|
The hook runs on the deployer machine, in your shell, not in a
|
|
298
341
|
container. So a token that only exists in `.kamal/secrets` for the
|
|
@@ -313,8 +356,8 @@ it exits 0 regardless.
|
|
|
313
356
|
|
|
314
357
|
## Log search finds less than it should
|
|
315
358
|
|
|
316
|
-
**Symptom.** On a
|
|
317
|
-
fewer lines and highlights nothing.
|
|
359
|
+
**Symptom.** On a self-hosted platform backed by Postgres, log search
|
|
360
|
+
matches fewer lines and highlights nothing.
|
|
318
361
|
|
|
319
362
|
**Cause.** Full-text search uses SQLite's FTS5 (`logs_fts`). The
|
|
320
363
|
platform checks for both a SQLite adapter *and* the `logs_fts` table.
|
|
@@ -9,10 +9,11 @@ module Railwatch
|
|
|
9
9
|
source_root File.expand_path("templates", __dir__)
|
|
10
10
|
|
|
11
11
|
desc "Creates config/initializers/railwatch.rb, a Kamal post-deploy hook, the browser client, and wires the test helpers. " \
|
|
12
|
-
"
|
|
12
|
+
"By default telemetry stays in this app, in two SQLite databases, with the dashboard at /railwatch. " \
|
|
13
|
+
"--cloud (or any token or URL option) sends it to Railwatch Cloud instead."
|
|
13
14
|
|
|
14
|
-
class_option :
|
|
15
|
-
desc: "
|
|
15
|
+
class_option :cloud, type: :boolean, default: false,
|
|
16
|
+
desc: "Send telemetry to Railwatch Cloud instead of keeping it in this app. Implied by --prompt-token, --token-stdin, --url and --kamal-secrets."
|
|
16
17
|
class_option :prompt_token, type: :boolean, default: false,
|
|
17
18
|
desc: "Prompt for the ingest token without echoing it."
|
|
18
19
|
class_option :token_stdin, type: :boolean, default: false,
|
|
@@ -67,9 +68,9 @@ module Railwatch
|
|
|
67
68
|
# database is, so an app on PostgreSQL or MySQL needs the adapter gem
|
|
68
69
|
# added before those files can be created.
|
|
69
70
|
def ensure_sqlite3_gem
|
|
70
|
-
return unless
|
|
71
|
+
return unless local?
|
|
71
72
|
return if Gem.loaded_specs.key?("sqlite3")
|
|
72
|
-
return say("
|
|
73
|
+
return say("Embedded mode needs the sqlite3 gem for its two databases; add `gem \"sqlite3\"` and re-run.", :yellow) unless File.exist?("Gemfile")
|
|
73
74
|
|
|
74
75
|
contents = File.read("Gemfile")
|
|
75
76
|
unless contents.match?(/^\s*gem ["']sqlite3["']/)
|
|
@@ -89,9 +90,9 @@ module Railwatch
|
|
|
89
90
|
# creates the tables now and migrates them after every gem update.
|
|
90
91
|
# Nothing is copied into the app.
|
|
91
92
|
def configure_local_databases
|
|
92
|
-
return unless
|
|
93
|
+
return unless local?
|
|
93
94
|
|
|
94
|
-
return say("
|
|
95
|
+
return say("No config/database.yml found; add railwatch and railwatch_telemetry databases yourself (docs/embedded.md).", :yellow) unless File.exist?("config/database.yml")
|
|
95
96
|
|
|
96
97
|
contents = File.read("config/database.yml")
|
|
97
98
|
updated = self.class.database_yml_with_railwatch(contents)
|
|
@@ -103,8 +104,8 @@ module Railwatch
|
|
|
103
104
|
# The writer process: one per Puma master, forked by the gem's Puma
|
|
104
105
|
# plugin, so batches are mapped and written outside the web workers.
|
|
105
106
|
def configure_local_writer
|
|
106
|
-
return unless
|
|
107
|
-
return say("
|
|
107
|
+
return unless local?
|
|
108
|
+
return say("No config/puma.rb found; add `plugin :railwatch` to your Puma config yourself (docs/embedded.md).", :yellow) unless File.exist?("config/puma.rb")
|
|
108
109
|
|
|
109
110
|
contents = File.read("config/puma.rb")
|
|
110
111
|
updated = self.class.puma_rb_with_railwatch(contents)
|
|
@@ -166,7 +167,7 @@ module Railwatch
|
|
|
166
167
|
# A token lands in .env only when Git confirms the file is ignored.
|
|
167
168
|
# URLs are not secret and can still be written to a tracked dotenv file.
|
|
168
169
|
def write_env
|
|
169
|
-
return if
|
|
170
|
+
return if local?
|
|
170
171
|
|
|
171
172
|
token = resolved_token
|
|
172
173
|
vars = { TOKEN_VAR => token, URL_VAR => options[:url] }.compact
|
|
@@ -219,7 +220,7 @@ module Railwatch
|
|
|
219
220
|
# the same prepare a deploy runs, for both databases only. The host's
|
|
220
221
|
# own databases are not touched, and a schema file is never written.
|
|
221
222
|
def prepare_local_databases
|
|
222
|
-
return unless
|
|
223
|
+
return unless local?
|
|
223
224
|
return unless File.exist?("config/database.yml")
|
|
224
225
|
return if @needs_bundle
|
|
225
226
|
return unless defined?(Rails) && Rails.respond_to?(:application) && Rails.application
|
|
@@ -242,21 +243,26 @@ module Railwatch
|
|
|
242
243
|
end
|
|
243
244
|
|
|
244
245
|
def show_next_steps
|
|
245
|
-
if
|
|
246
|
+
if local?
|
|
246
247
|
say <<~STEPS, :green
|
|
247
248
|
|
|
248
249
|
Next steps
|
|
249
|
-
1.
|
|
250
|
-
|
|
250
|
+
1. Restart the app and open /railwatch. In development it is open.
|
|
251
|
+
2. Before production, give it a password (it is closed there until
|
|
252
|
+
you do): RAILS_ENV=production bin/rails railwatch:authentication:configure
|
|
251
253
|
Using your own admin auth instead? See docs/embedded.md,
|
|
252
254
|
Authentication (base_controller_class or a routes constraint).
|
|
253
|
-
2. Restart the app and open /railwatch.
|
|
254
255
|
3. #{@prepared ? "Nothing else to run. Both databases were created just now and" : "Create the two databases: bin/rails db:prepare\n Then"}
|
|
255
256
|
`bin/rails db:prepare` (which a deploy already runs) migrates
|
|
256
257
|
them after every gem update. With `plugin :railwatch` in
|
|
257
258
|
config/puma.rb Puma forks one Railwatch writer process that
|
|
258
259
|
writes every batch and runs the maintenance clock, so no web
|
|
259
260
|
process ever holds the telemetry database. No job worker.
|
|
261
|
+
4. Optional: mirror to Railwatch Cloud for alerts delivered even
|
|
262
|
+
when this app is down, MCP for your AI assistant, and every
|
|
263
|
+
app in one place. Get a token (bin/rails railwatch:token), set
|
|
264
|
+
RAILWATCH_TOKEN, then c.export_enabled = true in the
|
|
265
|
+
initializer (docs/embedded.md).
|
|
260
266
|
STEPS
|
|
261
267
|
return
|
|
262
268
|
end
|
|
@@ -287,7 +293,7 @@ module Railwatch
|
|
|
287
293
|
# This process read its configuration before the initializer was
|
|
288
294
|
# written, so in local mode the doctor would report an http transport
|
|
289
295
|
# with no token. The databases it would check were prepared above.
|
|
290
|
-
return if
|
|
296
|
+
return if local?
|
|
291
297
|
return unless defined?(Rails) && Rails.respond_to?(:application) && Rails.application
|
|
292
298
|
|
|
293
299
|
say "\nbin/rails railwatch:doctor", :green
|
|
@@ -455,6 +461,16 @@ module Railwatch
|
|
|
455
461
|
|
|
456
462
|
private
|
|
457
463
|
|
|
464
|
+
# Embedded unless the invocation asks for the cloud: --cloud itself, or
|
|
465
|
+
# an option that only means something there. A RAILWATCH_TOKEN already
|
|
466
|
+
# in the environment is not asking -- it is picked up when one of these
|
|
467
|
+
# is given, never used to choose the mode.
|
|
468
|
+
CLOUD_OPTIONS = %i[cloud prompt_token token_stdin url kamal_secrets].freeze
|
|
469
|
+
|
|
470
|
+
def local?
|
|
471
|
+
CLOUD_OPTIONS.none? { |name| options[name] }
|
|
472
|
+
end
|
|
473
|
+
|
|
458
474
|
def resolved_token
|
|
459
475
|
@resolved_token ||= begin
|
|
460
476
|
value = if options[:prompt_token]
|
|
@@ -3,19 +3,19 @@
|
|
|
3
3
|
# Railwatch: first-class monitoring for Rails. Every option here can also be
|
|
4
4
|
# set by the RAILWATCH_* env var named in the comment.
|
|
5
5
|
Railwatch.configure do |c|
|
|
6
|
-
<% if
|
|
6
|
+
<% if local? -%>
|
|
7
7
|
# Telemetry stays in this app's own railwatch_telemetry database and the
|
|
8
|
-
# dashboard is served at /railwatch. No token, no cloud.
|
|
9
|
-
#
|
|
8
|
+
# dashboard is served at /railwatch. No token, no cloud. Who can open the
|
|
9
|
+
# dashboard is the Access block below.
|
|
10
10
|
c.transport = :local # RAILWATCH_TRANSPORT
|
|
11
11
|
c.ignored_request_paths += ["/railwatch", %r{\A/railwatch/}]
|
|
12
12
|
# c.issue_prefix = "APP" # RAILWATCH_ISSUE_PREFIX; issue keys like APP-12
|
|
13
13
|
# c.repository_url = "https://github.com/you/app" # RAILWATCH_REPOSITORY_URL; source links from stack traces
|
|
14
14
|
# c.retention_days = 7 # RAILWATCH_RETENTION_DAYS; PruneTelemetryJob keeps this much
|
|
15
15
|
# Access. The dashboard shows every query, log line and exception this app
|
|
16
|
-
# records, so pick one of these. Out of the box it is HTTP Basic
|
|
17
|
-
# closed until credentials exist:
|
|
18
|
-
# bin/rails railwatch:authentication:configure
|
|
16
|
+
# records, so pick one of these. Out of the box it is HTTP Basic: open in
|
|
17
|
+
# development, closed everywhere else until credentials exist:
|
|
18
|
+
# RAILS_ENV=production bin/rails railwatch:authentication:configure
|
|
19
19
|
# or RAILWATCH_HTTP_BASIC_AUTH_USER / _PASSWORD.
|
|
20
20
|
#
|
|
21
21
|
# Using your own admin auth instead? Turn Basic off and say which, so the
|
|
@@ -21,6 +21,11 @@ Puma::Plugin.create do
|
|
|
21
21
|
attr_reader :log_writer, :writer_pid
|
|
22
22
|
|
|
23
23
|
POLL = 2
|
|
24
|
+
# How often a stopping writer is checked for, and how long a KILL is given
|
|
25
|
+
# to take before the pid is abandoned (a process stuck in disk I/O cannot
|
|
26
|
+
# die until the I/O returns, and Puma's exit should not wait for that).
|
|
27
|
+
REAP_POLL = 0.05
|
|
28
|
+
KILL_REAP = 1
|
|
24
29
|
|
|
25
30
|
def start(launcher)
|
|
26
31
|
@log_writer = launcher.log_writer
|
|
@@ -151,19 +156,59 @@ Puma::Plugin.create do
|
|
|
151
156
|
log "Railwatch writer shutdown failed (#{e.class}: #{e.message})"
|
|
152
157
|
end
|
|
153
158
|
|
|
154
|
-
# TERM closes the writer's listener
|
|
155
|
-
#
|
|
159
|
+
# TERM closes the writer's listener; it finishes what it is holding and
|
|
160
|
+
# exits on its own. Waited for with a deadline, not Process.wait, which
|
|
161
|
+
# has none: a writer wedged in a SQLite write or on a full disk would hold
|
|
162
|
+
# Puma's exit open for as long as it stayed wedged. KILL past the deadline.
|
|
163
|
+
# Reaped either way, so a cluster master never leaves a zombie behind.
|
|
156
164
|
def stop_writer
|
|
157
165
|
return unless @writer_pid
|
|
158
166
|
|
|
159
167
|
Process.kill(:TERM, @writer_pid)
|
|
160
|
-
|
|
168
|
+
return if reaped_within?(stop_timeout)
|
|
169
|
+
|
|
170
|
+
log "Railwatch writer (pid #{@writer_pid}) did not exit within #{stop_timeout}s of TERM; killing it"
|
|
171
|
+
Process.kill(:KILL, @writer_pid)
|
|
172
|
+
log "Railwatch writer (pid #{@writer_pid}) did not exit on KILL; leaving it" unless reaped_within?(KILL_REAP)
|
|
161
173
|
rescue Errno::ECHILD, Errno::ESRCH
|
|
162
174
|
nil
|
|
163
175
|
ensure
|
|
164
176
|
@writer_pid = nil
|
|
165
177
|
end
|
|
166
178
|
|
|
179
|
+
# shutdown_timeout: the same allowance this process gives its own
|
|
180
|
+
# reporter, and deliberately NOT the writer's full theoretical exit time
|
|
181
|
+
# (a sequential SHUTDOWN_DRAIN join per worker thread, its maintenance
|
|
182
|
+
# join, then its own reporter shutdown -- 13s at the defaults). Waiting
|
|
183
|
+
# that long would buy nothing: a writer killed mid-batch loses no data,
|
|
184
|
+
# because the transaction rolls back and the worker retries the batch by
|
|
185
|
+
# id against the next writer (Writer#serve says so). Exit time spent
|
|
186
|
+
# waiting for the drain is spent for nothing. An idle writer is gone in
|
|
187
|
+
# well under a second either way.
|
|
188
|
+
#
|
|
189
|
+
# This runs from at_exit, inside the container's TERM-to-KILL grace. Under
|
|
190
|
+
# Kamal that is Docker's 10s default for a proxied role that has not set
|
|
191
|
+
# `stop_timeout`, or whatever `stop_timeout` says when it has; an app that
|
|
192
|
+
# wants the writer given longer can raise RAILWATCH_SHUTDOWN_TIMEOUT to
|
|
193
|
+
# match its grace, and this bound rises with it.
|
|
194
|
+
def stop_timeout
|
|
195
|
+
[ ::Railwatch.config.shutdown_timeout.to_f, 0.0 ].max
|
|
196
|
+
end
|
|
197
|
+
|
|
198
|
+
# Non-blocking waits on a short poll; Process.wait has no timeout and a
|
|
199
|
+
# child that never exits would hold it forever. ECHILD (the cluster's
|
|
200
|
+
# wait2(-1) reaped it first) propagates to stop_writer, which reads it as
|
|
201
|
+
# gone.
|
|
202
|
+
def reaped_within?(seconds)
|
|
203
|
+
deadline = Process.clock_gettime(Process::CLOCK_MONOTONIC) + seconds
|
|
204
|
+
loop do
|
|
205
|
+
return true if Process.waitpid(@writer_pid, Process::WNOHANG)
|
|
206
|
+
return false if Process.clock_gettime(Process::CLOCK_MONOTONIC) >= deadline
|
|
207
|
+
|
|
208
|
+
sleep REAP_POLL
|
|
209
|
+
end
|
|
210
|
+
end
|
|
211
|
+
|
|
167
212
|
def log(message)
|
|
168
213
|
log_writer.log(message)
|
|
169
214
|
end
|
|
@@ -81,7 +81,7 @@ module Railwatch
|
|
|
81
81
|
:max_view_renders_per_execution, :ignored_cache_key_prefixes,
|
|
82
82
|
:beacon_enabled, :beacon_rate_limit, :beacon_global_rate_limit, :beacon_allowed_origins,
|
|
83
83
|
:debug, :capture_default_vendor_commands,
|
|
84
|
-
:capture_default_vendor_cache_keys, :on_unrecoverable,
|
|
84
|
+
:capture_default_vendor_cache_keys, :on_unrecoverable, :warn_on_data_loss,
|
|
85
85
|
:capture_framework_events,
|
|
86
86
|
:tail_sample_slow_ms, :failure_context, :propagate_traces, :trace_propagation_hosts,
|
|
87
87
|
:health_interval, :capture_query_explain, :explain_threshold_ms,
|
|
@@ -195,6 +195,12 @@ module Railwatch
|
|
|
195
195
|
@capture_default_vendor_cache_keys = env_bool("RAILWATCH_CAPTURE_DEFAULT_VENDOR_CACHE_KEYS", false)
|
|
196
196
|
@capture_framework_events = env_bool("RAILWATCH_CAPTURE_FRAMEWORK_EVENTS", false)
|
|
197
197
|
@on_unrecoverable = nil
|
|
198
|
+
# Off, because a gem printing into an application's own output is the
|
|
199
|
+
# gem changing that application's behaviour, and this one stays
|
|
200
|
+
# additive. Turn it on and a batch lost for good says so in one stderr
|
|
201
|
+
# line; leave it off and the loss shows behind RAILWATCH_DEBUG, or
|
|
202
|
+
# wherever on_unrecoverable routes it.
|
|
203
|
+
@warn_on_data_loss = env_bool("RAILWATCH_WARN_ON_DATA_LOSS", true)
|
|
198
204
|
@beacon_enabled = env_bool("RAILWATCH_BEACON", true)
|
|
199
205
|
# The beacon is unauthenticated and forces Railwatch.keep! for browser
|
|
200
206
|
# errors, so without a ceiling anyone can spend an app's event quota
|
|
@@ -372,12 +378,21 @@ module Railwatch
|
|
|
372
378
|
http_basic_auth_user.to_s.strip != "" && http_basic_auth_password.to_s.strip != ""
|
|
373
379
|
end
|
|
374
380
|
|
|
381
|
+
# Development with Basic on and nothing configured is open, so a first
|
|
382
|
+
# run is `rails g railwatch:install` and a page, not a password step
|
|
383
|
+
# first. Rails already shows full error pages there. Every other
|
|
384
|
+
# environment stays closed until credentials exist.
|
|
385
|
+
def http_basic_auth_waived?
|
|
386
|
+
http_basic_auth_enabled && !http_basic_auth_configured? && defined?(Rails) && Rails.env.development?
|
|
387
|
+
end
|
|
388
|
+
|
|
375
389
|
# Whether the request carries the configured HTTP Basic credentials.
|
|
376
390
|
# False when Basic is on and nothing is configured (closed), true when
|
|
377
391
|
# Basic is off (the host's base controller or routes constraint is the
|
|
378
392
|
# gate then). Shared by the dashboard controller and the live channel.
|
|
379
393
|
def http_basic_auth_ok?(request)
|
|
380
394
|
return true unless http_basic_auth_enabled
|
|
395
|
+
return true if http_basic_auth_waived?
|
|
381
396
|
return false unless http_basic_auth_configured?
|
|
382
397
|
|
|
383
398
|
ActionController::HttpAuthentication::Basic.authenticate(request) do |user, password|
|
data/lib/railwatch/engine.rb
CHANGED
|
@@ -91,16 +91,27 @@ module Railwatch
|
|
|
91
91
|
config.http_basic_auth_password ||= app.credentials.dig(:railwatch, :http_basic_auth_password)
|
|
92
92
|
end
|
|
93
93
|
|
|
94
|
-
#
|
|
94
|
+
# Three things worth one line in the log at boot, because all are
|
|
95
95
|
# invisible until something is already wrong: a Rails/json pair that
|
|
96
|
-
# cannot decode,
|
|
97
|
-
#
|
|
96
|
+
# cannot decode, an embedded dashboard with nothing declared in front of
|
|
97
|
+
# it, and one that is closed to everyone because Basic has no
|
|
98
|
+
# credentials outside development.
|
|
98
99
|
initializer "railwatch.warnings", after: :load_config_initializers do
|
|
99
100
|
config.after_initialize do
|
|
100
101
|
next unless Railwatch.enabled?
|
|
101
102
|
|
|
102
103
|
Rails.logger.warn("[railwatch] #{Railwatch::JsonCompat.advice}") if Railwatch::JsonCompat.broken?
|
|
103
104
|
|
|
105
|
+
if Railwatch.config.local? && Railwatch.config.dashboard_gate == :basic &&
|
|
106
|
+
!Railwatch.config.http_basic_auth_configured? && !Rails.env.local?
|
|
107
|
+
Rails.logger.warn(
|
|
108
|
+
"[railwatch] the dashboard is closed: HTTP Basic is on and no credentials are configured for " \
|
|
109
|
+
"#{Rails.env}, so every request to it is 401. Run `RAILS_ENV=#{Rails.env} bin/rails " \
|
|
110
|
+
"railwatch:authentication:configure`, or gate it with your own auth and set " \
|
|
111
|
+
"`c.http_basic_auth_enabled = false` (docs/embedded.md)."
|
|
112
|
+
)
|
|
113
|
+
end
|
|
114
|
+
|
|
104
115
|
if Railwatch.config.local? && Railwatch.config.dashboard_gate == :undeclared && !Rails.env.local?
|
|
105
116
|
Rails.logger.warn(
|
|
106
117
|
"[railwatch] the dashboard at the engine's mount has no gate this gem can see: HTTP Basic is off and " \
|
data/lib/railwatch/reporter.rb
CHANGED
|
@@ -100,10 +100,17 @@ module Railwatch
|
|
|
100
100
|
|
|
101
101
|
def flush
|
|
102
102
|
ensure_process!
|
|
103
|
-
|
|
103
|
+
deferred = nil
|
|
104
|
+
result = @flush_mutex.synchronize do
|
|
105
|
+
@deferred_notifications = []
|
|
104
106
|
update_backpressure
|
|
105
107
|
deliver_buffer
|
|
108
|
+
ensure
|
|
109
|
+
deferred = @deferred_notifications
|
|
110
|
+
@deferred_notifications = nil
|
|
106
111
|
end
|
|
112
|
+
deferred&.each { |error| Railwatch.notify_unrecoverable(error) }
|
|
113
|
+
result
|
|
107
114
|
end
|
|
108
115
|
|
|
109
116
|
def ensure_thread
|
|
@@ -278,6 +285,19 @@ module Railwatch
|
|
|
278
285
|
end
|
|
279
286
|
end
|
|
280
287
|
|
|
288
|
+
# Everything reachable from deliver_buffer runs inside @flush_mutex, so
|
|
289
|
+
# the loss it reports cannot be handed to the application there. Ruby's
|
|
290
|
+
# Mutex is not reentrant: the documented callback is Rails.error.report,
|
|
291
|
+
# whose subscriber records the error as an exception and can end up asking
|
|
292
|
+
# this same reporter to flush -- "ThreadError: deadlock; recursive
|
|
293
|
+
# locking", rescued by notify_unrecoverable and so a callback cut off
|
|
294
|
+
# halfway, reporting nothing. Collected here and dispatched by flush once
|
|
295
|
+
# the lock is released. retain already does this for @mutex; @flush_mutex
|
|
296
|
+
# is the outer one it still sat inside.
|
|
297
|
+
def defer_notification(error)
|
|
298
|
+
@deferred_notifications ? @deferred_notifications << error : Railwatch.notify_unrecoverable(error)
|
|
299
|
+
end
|
|
300
|
+
|
|
281
301
|
def deliver_buffer
|
|
282
302
|
# A 401 was reported once, when the transport first saw it; after
|
|
283
303
|
# that the token is wrong until the process restarts, and repeating
|
|
@@ -337,7 +357,7 @@ module Railwatch
|
|
|
337
357
|
rescue StandardError => e
|
|
338
358
|
result = Transport::Http::Result.new(ok: false, error: "#{e.class}: #{e.message}")
|
|
339
359
|
batch&.records&.any? ? retain(batch, result) : delivery_succeeded
|
|
340
|
-
|
|
360
|
+
defer_notification(e)
|
|
341
361
|
result
|
|
342
362
|
ensure
|
|
343
363
|
in_flight(0, 0, 0, 0)
|
|
@@ -350,7 +370,7 @@ module Railwatch
|
|
|
350
370
|
end
|
|
351
371
|
|
|
352
372
|
def retain(batch, result)
|
|
353
|
-
@mutex.synchronize do
|
|
373
|
+
gave_up = @mutex.synchronize do
|
|
354
374
|
@in_flight_records = 0
|
|
355
375
|
@in_flight_dropped = 0
|
|
356
376
|
@in_flight_bytes = 0
|
|
@@ -362,17 +382,7 @@ module Railwatch
|
|
|
362
382
|
@retry_attempt = 0
|
|
363
383
|
@retry_at = nil
|
|
364
384
|
Railwatch.debug { "gave up on a batch of #{batch.records.size} records after #{MAX_RETRY_ATTEMPTS} retries (#{result.error || result.status}); dropped and counted" }
|
|
365
|
-
|
|
366
|
-
# embedded writer) that never comes back this is the only place the
|
|
367
|
-
# loss is ever reported, and the dropped counter it leaves behind
|
|
368
|
-
# rides on the NEXT successful delivery, which may never happen.
|
|
369
|
-
Railwatch.notify_unrecoverable(
|
|
370
|
-
DeliveryError.new("Railwatch dropped #{batch.records.size} records after #{MAX_RETRY_ATTEMPTS} failed delivery attempts: " \
|
|
371
|
-
"#{result.error || result.status}",
|
|
372
|
-
status: result.status, records: batch.records.size, bytes: batch.bytes,
|
|
373
|
-
dropped: batch.dropped, dropped_bytes: batch.dropped_bytes)
|
|
374
|
-
)
|
|
375
|
-
next
|
|
385
|
+
next true
|
|
376
386
|
end
|
|
377
387
|
@retry_batch = batch
|
|
378
388
|
delay = retry_delay(@retry_attempt)
|
|
@@ -382,7 +392,32 @@ module Railwatch
|
|
|
382
392
|
"retained #{batch.records.size} records after retryable delivery failure " \
|
|
383
393
|
"(#{result.error || result.status}); retry #{@retry_attempt} in #{delay.round(3)}s"
|
|
384
394
|
end
|
|
395
|
+
false
|
|
385
396
|
end
|
|
397
|
+
return unless gave_up
|
|
398
|
+
|
|
399
|
+
# Losing a batch is not a debug-level event: with an ingest (or an
|
|
400
|
+
# embedded writer) that never comes back this is the only place the
|
|
401
|
+
# loss is ever reported, and the dropped counter it leaves behind
|
|
402
|
+
# rides on the NEXT successful delivery, which may never happen.
|
|
403
|
+
#
|
|
404
|
+
# Reported here, after the lock is released, never inside it. The
|
|
405
|
+
# callback is the app's: the documented one is Rails.error.report,
|
|
406
|
+
# whose subscriber records the error as an exception and so writes
|
|
407
|
+
# straight back into this reporter -- which needs @mutex to arm the
|
|
408
|
+
# thread or ask for a flush, and under the lock that was
|
|
409
|
+
# "ThreadError: deadlock; recursive locking" and a callback cut off
|
|
410
|
+
# halfway. A slow callback under the lock was worse: shutdown's own
|
|
411
|
+
# @mutex.synchronize and every request thread's write_now sat behind
|
|
412
|
+
# it for as long as it took. notify_unsent and delivery_rejected
|
|
413
|
+
# already call out unlocked; this was the one that did not.
|
|
414
|
+
defer_notification(
|
|
415
|
+
DeliveryError.new("Railwatch dropped #{batch.records.size} #{batch.records.size == 1 ? "record" : "records"} " \
|
|
416
|
+
"after #{MAX_RETRY_ATTEMPTS + 1} failed delivery attempts: " \
|
|
417
|
+
"#{result.error || result.status}",
|
|
418
|
+
status: result.status, records: batch.records.size, bytes: batch.bytes,
|
|
419
|
+
dropped: batch.dropped, dropped_bytes: batch.dropped_bytes)
|
|
420
|
+
)
|
|
386
421
|
end
|
|
387
422
|
|
|
388
423
|
def discard_unauthorized
|
|
@@ -406,7 +441,7 @@ module Railwatch
|
|
|
406
441
|
def delivery_rejected(batch, result)
|
|
407
442
|
delivery_succeeded
|
|
408
443
|
detail = result.error.to_s.empty? ? "HTTP #{result.status}" : result.error
|
|
409
|
-
|
|
444
|
+
defer_notification(
|
|
410
445
|
DeliveryError.new("Railwatch ingest permanently rejected #{batch.records.size} records: #{detail}",
|
|
411
446
|
status: result.status, records: batch.records.size, bytes: batch.bytes,
|
|
412
447
|
dropped: batch.dropped, dropped_bytes: batch.dropped_bytes)
|
|
@@ -550,7 +585,8 @@ module Railwatch
|
|
|
550
585
|
return unless should_notify
|
|
551
586
|
|
|
552
587
|
Railwatch.notify_unrecoverable(
|
|
553
|
-
DeliveryError.new("Railwatch #{reason} with #{records} unsent records
|
|
588
|
+
DeliveryError.new("Railwatch #{reason} with #{records} unsent #{records == 1 ? "record" : "records"} " \
|
|
589
|
+
"retained in memory (#{bytes} bytes)",
|
|
554
590
|
records: records, bytes: bytes, dropped: dropped, dropped_bytes: dropped_bytes)
|
|
555
591
|
)
|
|
556
592
|
end
|
|
@@ -8,9 +8,10 @@ require "json"
|
|
|
8
8
|
|
|
9
9
|
module Railwatch
|
|
10
10
|
module Transport
|
|
11
|
-
# POSTs gzip NDJSON batches to the platform. Each call
|
|
12
|
-
#
|
|
13
|
-
#
|
|
11
|
+
# POSTs gzip NDJSON batches to the platform. Each call makes exactly one
|
|
12
|
+
# attempt and returns a classified, non-raising result; Reporter owns
|
|
13
|
+
# retention, the retry ladder, and backoff between calls, so a retry here
|
|
14
|
+
# would multiply into its schedule rather than add to it. A 401 marks the
|
|
14
15
|
# transport unauthorized so no further requests are made.
|
|
15
16
|
class Http
|
|
16
17
|
# The request headers a delivery carries besides the body. The receiver
|
|
@@ -87,17 +88,11 @@ module Railwatch
|
|
|
87
88
|
dropped += over_cap
|
|
88
89
|
dropped_bytes += over_cap_bytes
|
|
89
90
|
end
|
|
90
|
-
attempt = 0
|
|
91
91
|
begin
|
|
92
|
-
attempt += 1
|
|
93
92
|
result = parse(post(body, dropped, dropped_bytes, backpressure_factor, batch_id), expected_count: sent)
|
|
94
|
-
if attempt < 2 && (500..599).cover?(result.status)
|
|
95
|
-
result = parse(post(body, dropped, dropped_bytes, backpressure_factor, batch_id), expected_count: sent)
|
|
96
|
-
end
|
|
97
93
|
apply_status_policy(result)
|
|
98
94
|
result
|
|
99
95
|
rescue StandardError => e
|
|
100
|
-
retry if attempt < 2
|
|
101
96
|
Result.new(ok: false, error: "#{e.class}: #{e.message}")
|
|
102
97
|
end
|
|
103
98
|
end
|
data/lib/railwatch/version.rb
CHANGED